Coverage Report

Created: 2026-08-14 21:19

/root/doris/cloud/src/recycler/recycler.h
Line
Count
Source (jump to first uncovered line)
1
// Licensed to the Apache Software Foundation (ASF) under one
2
// or more contributor license agreements.  See the NOTICE file
3
// distributed with this work for additional information
4
// regarding copyright ownership.  The ASF licenses this file
5
// to you under the Apache License, Version 2.0 (the
6
// "License"); you may not use this file except in compliance
7
// with the License.  You may obtain a copy of the License at
8
//
9
//   http://www.apache.org/licenses/LICENSE-2.0
10
//
11
// Unless required by applicable law or agreed to in writing,
12
// software distributed under the License is distributed on an
13
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
14
// KIND, either express or implied.  See the License for the
15
// specific language governing permissions and limitations
16
// under the License.
17
18
#pragma once
19
20
#include <gen_cpp/cloud.pb.h>
21
#include <glog/logging.h>
22
23
#include <atomic>
24
#include <condition_variable>
25
#include <cstdint>
26
#include <deque>
27
#include <functional>
28
#include <memory>
29
#include <string>
30
#include <string_view>
31
#include <thread>
32
#include <unordered_map>
33
#include <unordered_set>
34
#include <utility>
35
36
#include "common/bvars.h"
37
#include "meta-service/delete_bitmap_lock_white_list.h"
38
#include "meta-service/txn_lazy_committer.h"
39
#include "meta-store/versionstamp.h"
40
#include "recycler/snapshot_chain_compactor.h"
41
#include "recycler/snapshot_data_migrator.h"
42
#include "recycler/storage_vault_accessor.h"
43
#include "snapshot/snapshot_manager.h"
44
45
namespace brpc {
46
class Server;
47
} // namespace brpc
48
49
namespace doris::cloud {
50
class TxnKv;
51
class InstanceRecycler;
52
class StorageVaultAccessor;
53
class Checker;
54
class SimpleThreadPool;
55
class RecyclerMetricsContext;
56
class TabletRecyclerMetricsContext;
57
class SegmentRecyclerMetricsContext;
58
59
int64_t calculate_tmp_rowset_expired_time(
60
        const std::string& instance_id_, const doris::RowsetMetaCloudPB& tmp_rowset_meta_pb,
61
        int64_t* earlest_ts /* tmp_rowset earliest expiration ts */);
62
63
struct RecyclerThreadPoolGroup {
64
11
    RecyclerThreadPoolGroup() = default;
65
    RecyclerThreadPoolGroup(std::shared_ptr<SimpleThreadPool> s3_producer_pool,
66
                            std::shared_ptr<SimpleThreadPool> recycle_tablet_pool,
67
                            std::shared_ptr<SimpleThreadPool> group_recycle_function_pool)
68
            : s3_producer_pool(std::move(s3_producer_pool)),
69
              recycle_tablet_pool(std::move(recycle_tablet_pool)),
70
13
              group_recycle_function_pool(std::move(group_recycle_function_pool)) {}
71
302
    ~RecyclerThreadPoolGroup() = default;
72
139
    RecyclerThreadPoolGroup(const RecyclerThreadPoolGroup&) = default;
73
    RecyclerThreadPoolGroup& operator=(RecyclerThreadPoolGroup& other) = default;
74
14
    RecyclerThreadPoolGroup& operator=(RecyclerThreadPoolGroup&& other) = default;
75
139
    RecyclerThreadPoolGroup(RecyclerThreadPoolGroup&&) = default;
76
    // used for accessor.delete_files, accessor.delete_directory
77
    std::shared_ptr<SimpleThreadPool> s3_producer_pool;
78
    // used for InstanceRecycler::recycle_tablet
79
    std::shared_ptr<SimpleThreadPool> recycle_tablet_pool;
80
    std::shared_ptr<SimpleThreadPool> group_recycle_function_pool;
81
};
82
83
class Recycler {
84
public:
85
    explicit Recycler(std::shared_ptr<TxnKv> txn_kv);
86
    ~Recycler();
87
88
    // returns 0 for success otherwise error
89
    int start(brpc::Server* server);
90
91
    void stop();
92
93
4.37k
    bool stopped() const { return stopped_.load(std::memory_order_acquire); }
94
95
0
    RecyclerThreadPoolGroup& thread_pool_group() { return _thread_pool_group; }
96
97
private:
98
    void recycle_callback();
99
100
    void instance_scanner_callback();
101
102
    void lease_recycle_jobs();
103
104
    void check_recycle_tasks();
105
106
private:
107
    friend class RecyclerServiceImpl;
108
109
    std::shared_ptr<TxnKv> txn_kv_;
110
    std::atomic_bool stopped_ {false};
111
112
    std::vector<std::thread> workers_;
113
114
    std::mutex mtx_;
115
    // notify recycle workers
116
    std::condition_variable pending_instance_cond_;
117
    std::deque<InstanceInfoPB> pending_instance_queue_;
118
    std::unordered_set<std::string> pending_instance_set_;
119
    std::unordered_map<std::string, std::shared_ptr<InstanceRecycler>> recycling_instance_map_;
120
    // notify instance scanner and lease thread
121
    std::condition_variable notifier_;
122
123
    std::string ip_port_;
124
125
    std::unique_ptr<Checker> checker_;
126
127
    RecyclerThreadPoolGroup _thread_pool_group;
128
129
    std::shared_ptr<TxnLazyCommitter> txn_lazy_committer_;
130
    std::shared_ptr<SnapshotManager> snapshot_manager_;
131
    std::shared_ptr<SnapshotDataMigrator> snapshot_data_migrator_;
132
    std::shared_ptr<SnapshotChainCompactor> snapshot_chain_compactor_;
133
};
134
135
enum class RowsetRecyclingState {
136
    FORMAL_ROWSET,
137
    TMP_ROWSET,
138
};
139
140
// Represents a single rowset deletion task for batch delete
141
struct RowsetDeleteTask {
142
    RowsetMetaCloudPB rowset_meta;
143
    std::string recycle_rowset_key;       // Primary key marking "pending recycle"
144
    std::string non_versioned_rowset_key; // Legacy non-versioned rowset meta key
145
    std::string versioned_rowset_key;     // Versioned meta rowset key
146
    Versionstamp versionstamp;
147
    std::string rowset_ref_count_key;
148
};
149
150
class RecyclerMetricsContext {
151
public:
152
11
    RecyclerMetricsContext() = default;
153
154
    RecyclerMetricsContext(std::string instance_id, std::string operation_type)
155
520
            : operation_type(std::move(operation_type)), instance_id(std::move(instance_id)) {
156
520
        start();
157
520
    }
158
159
531
    ~RecyclerMetricsContext() = default;
160
161
    std::atomic_ullong total_need_recycle_data_size = 0;
162
    std::atomic_ullong total_need_recycle_num = 0;
163
164
    std::atomic_ullong total_recycled_data_size = 0;
165
    std::atomic_ullong total_recycled_num = 0;
166
167
    std::string operation_type;
168
    std::string instance_id;
169
170
    double start_time = 0;
171
172
521
    void start() {
173
521
        start_time = duration_cast<std::chrono::milliseconds>(
174
521
                             std::chrono::system_clock::now().time_since_epoch())
175
521
                             .count();
176
521
    }
177
178
250
    double duration() const {
179
250
        return duration_cast<std::chrono::milliseconds>(
180
250
                       std::chrono::system_clock::now().time_since_epoch())
181
250
                       .count() -
182
250
               start_time;
183
250
    }
184
185
22
    void reset() {
186
22
        total_need_recycle_data_size = 0;
187
22
        total_need_recycle_num = 0;
188
22
        total_recycled_data_size = 0;
189
22
        total_recycled_num = 0;
190
22
        start_time = duration_cast<std::chrono::milliseconds>(
191
22
                             std::chrono::system_clock::now().time_since_epoch())
192
22
                             .count();
193
22
    }
194
195
250
    void finish_report() {
196
250
        if (!operation_type.empty()) {
197
250
            double cost = duration();
198
250
            g_bvar_recycler_instance_last_round_recycle_elpased_ts.put(
199
250
                    {instance_id, operation_type}, cost);
200
250
            g_bvar_recycler_instance_recycle_round.put({instance_id, operation_type}, 1);
201
250
            g_bvar_recycler_instance_recycle_total_bytes_since_started.put(
202
250
                    {instance_id, operation_type}, total_recycled_data_size.load());
203
250
            g_bvar_recycler_instance_recycle_total_num_since_started.put(
204
250
                    {instance_id, operation_type}, total_recycled_num.load());
205
250
            LOG(INFO) << "recycle instance: " << instance_id
206
250
                      << ", operation type: " << operation_type << ", cost: " << cost
207
250
                      << " ms, total recycled num: " << total_recycled_num.load()
208
250
                      << ", total recycled data size: " << total_recycled_data_size.load()
209
250
                      << " bytes";
210
250
            if (cost != 0) {
211
222
                if (total_recycled_num.load() != 0) {
212
56
                    g_bvar_recycler_instance_recycle_time_per_resource.put(
213
56
                            {instance_id, operation_type}, cost / total_recycled_num.load());
214
56
                }
215
222
                g_bvar_recycler_instance_recycle_bytes_per_ms.put(
216
222
                        {instance_id, operation_type}, total_recycled_data_size.load() / cost);
217
222
            }
218
250
        }
219
250
    }
220
221
    // `is_begin` is used to initialize total num of items need to be recycled
222
1.07k
    void report(bool is_begin = false) {
223
1.07k
        if (!operation_type.empty()) {
224
            // is init
225
1.04k
            if (is_begin) {
226
3
                auto value = total_need_recycle_num.load();
227
228
3
                g_bvar_recycler_instance_last_round_to_recycle_bytes.put(
229
3
                        {instance_id, operation_type}, total_need_recycle_data_size.load());
230
3
                g_bvar_recycler_instance_last_round_to_recycle_num.put(
231
3
                        {instance_id, operation_type}, value);
232
1.04k
            } else {
233
1.04k
                g_bvar_recycler_instance_last_round_recycled_bytes.put(
234
1.04k
                        {instance_id, operation_type}, total_recycled_data_size.load());
235
1.04k
                g_bvar_recycler_instance_last_round_recycled_num.put({instance_id, operation_type},
236
1.04k
                                                                     total_recycled_num.load());
237
1.04k
            }
238
1.04k
        }
239
1.07k
    }
240
};
241
242
class TabletRecyclerMetricsContext : public RecyclerMetricsContext {
243
public:
244
139
    TabletRecyclerMetricsContext() : RecyclerMetricsContext("global_recycler", "recycle_tablet") {}
245
};
246
247
class SegmentRecyclerMetricsContext : public RecyclerMetricsContext {
248
public:
249
    SegmentRecyclerMetricsContext()
250
139
            : RecyclerMetricsContext("global_recycler", "recycle_segment") {}
251
};
252
253
struct OplogRecycleStats;
254
255
class InstanceRecycler {
256
public:
257
    struct PackedFileRecycleStats {
258
        int64_t num_scanned = 0;          // packed-file kv scanned
259
        int64_t num_corrected = 0;        // packed-file kv corrected
260
        int64_t num_deleted = 0;          // packed-file kv deleted
261
        int64_t num_failed = 0;           // packed-file kv failed
262
        int64_t bytes_deleted = 0;        // packed-file kv bytes deleted from txn-kv
263
        int64_t num_object_deleted = 0;   // packed-file objects deleted from storage (vault/HDFS)
264
        int64_t bytes_object_deleted = 0; // bytes deleted from storage objects
265
        int64_t rowset_scan_count = 0;    // rowset metas scanned during correction
266
    };
267
268
    explicit InstanceRecycler(std::shared_ptr<TxnKv> txn_kv, const InstanceInfoPB& instance,
269
                              RecyclerThreadPoolGroup thread_pool_group,
270
                              std::shared_ptr<TxnLazyCommitter> txn_lazy_committer);
271
    ~InstanceRecycler();
272
273
0
    std::string_view instance_id() const { return instance_id_; }
274
9
    const InstanceInfoPB& instance_info() const { return instance_info_; }
275
276
    // returns 0 for success otherwise error
277
    int init();
278
279
0
    void stop() { stopped_.store(true, std::memory_order_release); }
280
56
    bool stopped() const { return stopped_.load(std::memory_order_acquire); }
281
282
    // returns 0 for success otherwise error
283
    int do_recycle();
284
285
    // remove all kv and data in this instance, ONLY be called when instance has been deleted
286
    // returns 0 for success otherwise error
287
    int recycle_deleted_instance();
288
289
    int recycle_deleted_instance_data();
290
291
    int recycle_deleted_instance_metadata();
292
293
    int update_instance_recycle_state(InstanceRecycleState expected_state,
294
                                      InstanceRecycleState target_state);
295
296
    int update_instance_recycle_state(InstanceRecycleState expected_state,
297
                                      InstanceRecycleState target_state, Transaction* txn);
298
299
    // scan and recycle expired indexes:
300
    // 1. dropped table, dropped mv
301
    // 2. half-successtable/index when create
302
    // returns 0 for success otherwise error
303
    int recycle_indexes();
304
305
    // scan and recycle expired partitions:
306
    // 1. dropped parttion
307
    // 2. half-success partition when create
308
    // returns 0 for success otherwise error
309
    int recycle_partitions();
310
311
    // scan and recycle expired rowsets:
312
    // 1. prepare_rowset will produce recycle_rowset before uploading data to remote storage (memo)
313
    // 2. compaction will change the input rowsets to recycle_rowset
314
    // returns 0 for success otherwise error
315
    int recycle_rowsets();
316
317
    // like `recycle_rowsets`, but for versioned rowsets.
318
    int recycle_versioned_rowsets();
319
320
    // scan and recycle expired tmp rowsets:
321
    // 1. commit_rowset will produce tmp_rowset when finish upload data (load or compaction) to remote storage
322
    // returns 0 for success otherwise error
323
    int recycle_tmp_rowsets();
324
325
    /**
326
     * recycle all tablets belonging to the index specified by `index_id`
327
     *
328
     * @param partition_id if positive, only recycle tablets in this partition belonging to the specified index
329
     * @return 0 for success otherwise error
330
     */
331
    int recycle_tablets(int64_t table_id, int64_t index_id, RecyclerMetricsContext& ctx,
332
                        int64_t partition_id = -1);
333
334
    /**
335
     * recycle all rowsets belonging to the tablet specified by `tablet_id`
336
     *
337
     * @return 0 for success otherwise error
338
     */
339
    int recycle_tablet(int64_t tablet_id, RecyclerMetricsContext& metrics_context);
340
341
    /**
342
     * like `recycle_tablet`, but for versioned tablet
343
     */
344
    int recycle_versioned_tablet(int64_t tablet_id, RecyclerMetricsContext& metrics_context);
345
346
    // scan and recycle useless partition version kv
347
    int recycle_versions();
348
349
    // scan and recycle the orphan partitions
350
    int recycle_orphan_partitions();
351
352
    // scan and abort timeout txn label
353
    // returns 0 for success otherwise error
354
    int abort_timeout_txn();
355
356
    //scan and recycle expire txn label
357
    // returns 0 for success otherwise error
358
    int recycle_expired_txn_label();
359
360
    // scan and recycle finished or timeout copy jobs
361
    // returns 0 for success otherwise error
362
    int recycle_copy_jobs();
363
364
    // scan and recycle dropped internal stage
365
    // returns 0 for success otherwise error
366
    int recycle_stage();
367
368
    // scan and recycle expired stage objects
369
    // returns 0 for success otherwise error
370
    int recycle_expired_stage_objects();
371
372
    // scan and recycle operation logs
373
    // returns 0 for success otherwise error
374
    int recycle_operation_logs();
375
376
    // scan and recycle expired restore jobs
377
    // returns 0 for success otherwise error
378
    int recycle_restore_jobs();
379
380
    /**
381
     * Scan packed-file metadata, correct reference counters, and recycle unused packed files.
382
     *
383
     * @return 0 on success, non-zero error code otherwise
384
     */
385
    int recycle_packed_files();
386
387
    // scan and recycle snapshots
388
    // returns 0 for success otherwise error
389
    int recycle_cluster_snapshots();
390
391
    // scan and recycle ref rowsets for deleted instance
392
    // returns 0 for success otherwise error
393
    int recycle_ref_rowsets(bool* has_unrecycled_rowsets);
394
395
    bool check_recycle_tasks();
396
397
    int scan_and_statistics_indexes();
398
399
    int scan_and_statistics_partitions();
400
401
    int scan_and_statistics_rowsets();
402
403
    int scan_and_statistics_tmp_rowsets();
404
405
    int scan_and_statistics_abort_timeout_txn();
406
407
    int scan_and_statistics_expired_txn_label();
408
409
    int scan_and_statistics_copy_jobs();
410
411
    int scan_and_statistics_stage();
412
413
    int scan_and_statistics_expired_stage_objects();
414
415
    int scan_and_statistics_versions();
416
417
    int scan_and_statistics_restore_jobs();
418
419
    void scan_and_statistics_operation_logs();
420
421
    /**
422
     * Decode the key of a packed-file metadata record into the persisted object path.
423
     *
424
     * @param key raw key persisted in txn-kv
425
     * @param packed_path output object storage path referenced by the key
426
     * @return true if decoding succeeds, false otherwise
427
     */
428
    static bool decode_packed_file_key(std::string_view key, std::string* packed_path);
429
430
30
    void TEST_add_accessor(std::string_view id, std::shared_ptr<StorageVaultAccessor> accessor) {
431
30
        accessor_map_.insert({std::string(id), std::move(accessor)});
432
30
    }
433
434
    // Recycle snapshot meta and data, return 0 for success otherwise error.
435
    int recycle_snapshot_meta_and_data(const std::string& instance_id,
436
                                       const std::string& resource_id,
437
                                       Versionstamp snapshot_version,
438
                                       const SnapshotPB& snapshot_pb);
439
440
private:
441
    // returns 0 for success otherwise error
442
    int remove_instance_key();
443
444
    // returns 0 for success otherwise error
445
    int init_obj_store_accessors();
446
447
    // returns 0 for success otherwise error
448
    int init_storage_vault_accessors();
449
450
    /**
451
     * Scan key-value pairs between [`begin`, `end`) with multiple rounds of range get(`RangeGetIterator`),
452
     * and perform `recycle_func` on each key-value pair.
453
     *
454
     * @param recycle_func defines how to recycle resources corresponding to a key-value pair.
455
     *                     The scan will stop if recycle_func() returns non-zero.
456
     *                     recycle_func() returns 0 if the recycling is successful or the scan can continue with ignorable errors.
457
     * @param loop_done is called after a round (`RangeGetIterator`) in the scan has no next kv. Usually used to perform a batch recycling.
458
     *                  The scan will stop if loop_done() returns non-zero.
459
     *                  loop_done() returns 0 if the recycling is successful or the scan can continue with ignorable errors.
460
     * @return 0 if all corresponding resources are recycled successfully, otherwise non-zero
461
     */
462
    int scan_and_recycle(std::string begin, std::string_view end,
463
                         std::function<int(std::string_view k, std::string_view v)> recycle_func,
464
                         std::function<int()> loop_done = nullptr);
465
466
    // return 0 for success otherwise error
467
    int delete_rowset_data(const doris::RowsetMetaCloudPB& rs_meta_pb);
468
469
    // return 0 for success otherwise error
470
    // NOTE: this function ONLY be called when the file paths cannot be calculated
471
    int delete_rowset_data(const std::string& resource_id, int64_t tablet_id,
472
                           const std::string& rowset_id);
473
474
    // return 0 for success otherwise error
475
    int delete_rowset_data(const std::map<std::string, doris::RowsetMetaCloudPB>& rowsets,
476
                           RowsetRecyclingState type, RecyclerMetricsContext& metrics_context);
477
478
    // Decrement packed file ref counts for rowset segments.
479
    // Returns 0 for success, -1 for error.
480
    int decrement_packed_file_ref_counts(const doris::RowsetMetaCloudPB& rs_meta_pb);
481
482
    enum class DeleteBitmapStorageType {
483
        NOT_FOUND,
484
        IN_FDB,
485
        STANDALONE_FILE,
486
        PACKED_FILE,
487
    };
488
489
    // Process delete bitmap storage and decrement packed file ref count when needed.
490
    // Returns 0 for success, -1 for error.
491
    // out_storage_type: if not null, will be set to the delete bitmap storage type.
492
    int decrement_delete_bitmap_packed_file_ref_counts(int64_t tablet_id,
493
                                                       const std::string& rowset_id,
494
                                                       DeleteBitmapStorageType* out_storage_type);
495
496
    int delete_packed_file_and_kv(const std::string& packed_file_path,
497
                                  const std::string& packed_key,
498
                                  const cloud::PackedFileInfoPB& packed_info);
499
500
    /**
501
     * Get stage storage info from instance and init StorageVaultAccessor
502
     * @return 0 if accessor is successfully inited, 1 if stage not found, negative for error
503
     */
504
    int init_copy_job_accessor(const std::string& stage_id, const StagePB::StageType& stage_type,
505
                               std::shared_ptr<StorageVaultAccessor>* accessor);
506
507
    void register_recycle_task(const std::string& task_name, int64_t start_time);
508
509
    void unregister_recycle_task(const std::string& task_name);
510
511
    // for scan all tablets and statistics metrics
512
    int scan_tablets_and_statistics(int64_t tablet_id, int64_t index_id,
513
                                    RecyclerMetricsContext& metrics_context,
514
                                    int64_t partition_id = -1, bool is_empty_tablet = false);
515
516
    // for scan all rs of tablet and statistics metrics
517
    int scan_tablet_and_statistics(int64_t tablet_id, RecyclerMetricsContext& metrics_context);
518
519
    // Recycle operation log and the log keys. The log keys are specified by `raw_keys`.
520
    //
521
    // Both `operation_log` and `raw_keys` will be removed in the same transaction, to ensure atomicity.
522
    int recycle_operation_log(Versionstamp log_version, const std::vector<std::string>& raw_keys,
523
                              OperationLogPB operation_log,
524
                              OplogRecycleStats* oplog_stats = nullptr);
525
526
    // Recycle rowset meta and data, return 0 for success otherwise error
527
    //
528
    // This function will decrease the rowset ref count and remove the rowset meta and data if the ref count is 1.
529
    int recycle_rowset_meta_and_data(const RowsetDeleteTask& task);
530
531
    // Classify rowset task by ref_count, return 0 to add to batch delete, 1 if handled (ref>1), -1 on error
532
    int classify_rowset_task_by_ref_count(RowsetDeleteTask& task,
533
                                          std::vector<RowsetDeleteTask>& batch_delete_tasks);
534
535
    // Cleanup metadata for deleted rowsets, return 0 for success otherwise error
536
    int cleanup_rowset_metadata(const std::vector<RowsetDeleteTask>& tasks);
537
538
    // Whether the instance has any snapshots, return 0 for success otherwise error.
539
    int has_cluster_snapshots(bool* any);
540
541
    // Whether need to recycle versioned keys
542
    bool should_recycle_versioned_keys() const;
543
    /**
544
     * Parse the path of a packed-file fragment and output the owning tablet and rowset identifiers.
545
     *
546
     * @param path packed-file fragment path to decode
547
     * @param tablet_id output tablet identifier extracted from the path
548
     * @param rowset_id output rowset identifier extracted from the path
549
     * @return true if both identifiers are successfully parsed, false otherwise
550
     */
551
    static bool parse_packed_slice_path(std::string_view path, int64_t* tablet_id,
552
                                        std::string* rowset_id);
553
    // Check whether a rowset referenced by a packed file still exists in metadata.
554
    // @param stats optional recycle statistics collector.
555
    int check_rowset_exists(int64_t tablet_id, const std::string& rowset_id, bool* exists,
556
                            PackedFileRecycleStats* stats = nullptr);
557
    int check_recycle_and_tmp_rowset_exists(int64_t tablet_id, const std::string& rowset_id,
558
                                            int64_t txn_id, bool* recycle_exists, bool* tmp_exists);
559
    /**
560
     * Resolve which storage accessor should be used for a packed file.
561
     *
562
     * @param hint preferred storage resource identifier persisted with the file
563
     * @return pair of the resolved resource identifier and accessor; the accessor can be null if unavailable
564
     */
565
    std::pair<std::string, std::shared_ptr<StorageVaultAccessor>> resolve_packed_file_accessor(
566
            const std::string& hint);
567
    // Recompute packed-file counters and lifecycle state after validating contained fragments.
568
    // @param stats optional recycle statistics collector.
569
    int correct_packed_file_info(cloud::PackedFileInfoPB* packed_info, bool* changed,
570
                                 const std::string& packed_file_path,
571
                                 PackedFileRecycleStats* stats = nullptr);
572
    // Correct and recycle a single packed-file record, updating metadata and accounting statistics.
573
    // @param stats optional recycle statistics collector.
574
    int process_single_packed_file(const std::string& packed_key,
575
                                   const std::string& packed_file_path,
576
                                   PackedFileRecycleStats* stats);
577
    // Process a packed-file KV while scanning and aggregate recycling statistics.
578
    int handle_packed_file_kv(std::string_view key, std::string_view value,
579
                              PackedFileRecycleStats* stats, int* ret);
580
581
    // Abort the transaction/job associated with a rowset that is about to be recycled.
582
    // This function is called during rowset recycling to prevent data loss by ensuring that
583
    // the transaction/job cannot be committed after its rowset data has been deleted.
584
    //
585
    // Scenario:
586
    // When recycler detects an expired prepared rowset (e.g., from a failed load transaction/job),
587
    // it needs to recycle the rowset data. However, if the transaction/job is still active and gets
588
    // committed after the data is deleted, it would lead to data loss - the transaction/job would
589
    // reference non-existent data.
590
    //
591
    // Solution:
592
    // Before recycling the rowset data, this function aborts the associated transaction/job to ensure
593
    // it cannot be committed. This guarantees that:
594
    // 1. The transaction/job state is marked as ABORTED
595
    // 2. Any subsequent commit_rowset/commit_txn attempts will fail
596
    // 3. The rowset data can be safely deleted without risk of data loss
597
    //
598
    // Parameters:
599
    //   txn_id: The transaction/job ID associated with the rowset to be recycled
600
    //
601
    // Returns:
602
    //   0 on success, -1 on failure
603
    int abort_txn_for_related_rowset(int64_t txn_id);
604
    int abort_job_for_related_rowset(const RowsetMetaCloudPB& rowset_meta);
605
606
    template <typename T>
607
    int batch_abort_txn_or_job_for_recycle(const std::vector<std::string>& keys,
608
                                           bool skip_base_version);
609
610
private:
611
    std::atomic_bool stopped_ {false};
612
    std::shared_ptr<TxnKv> txn_kv_;
613
    std::string instance_id_;
614
    InstanceInfoPB instance_info_;
615
616
    // TODO(plat1ko): Add new accessor to map in runtime for new created storage vaults
617
    std::unordered_map<std::string, std::shared_ptr<StorageVaultAccessor>> accessor_map_;
618
    using InvertedIndexInfo =
619
            std::pair<InvertedIndexStorageFormatPB, std::vector<std::pair<int64_t, std::string>>>;
620
621
    class InvertedIndexIdCache;
622
    std::unique_ptr<InvertedIndexIdCache> inverted_index_id_cache_;
623
624
    std::mutex recycled_tablets_mtx_;
625
    // Store recycled tablets, we can skip deleting rowset data of these tablets because these data has already been deleted.
626
    std::unordered_set<int64_t> recycled_tablets_;
627
628
    std::mutex recycle_tasks_mutex;
629
    // <task_name, start_time>>
630
    std::map<std::string, int64_t> running_recycle_tasks;
631
632
    RecyclerThreadPoolGroup _thread_pool_group;
633
634
    std::shared_ptr<TxnLazyCommitter> txn_lazy_committer_;
635
    std::shared_ptr<SnapshotManager> snapshot_manager_;
636
    std::shared_ptr<DeleteBitmapLockWhiteList> delete_bitmap_lock_white_list_;
637
    std::shared_ptr<ResourceManager> resource_mgr_;
638
639
    TabletRecyclerMetricsContext tablet_metrics_context_;
640
    SegmentRecyclerMetricsContext segment_metrics_context_;
641
};
642
643
struct OperationLogReferenceInfo {
644
    bool referenced_by_instance = false;
645
    bool referenced_by_snapshot = false;
646
    Versionstamp referenced_snapshot_timestamp;
647
};
648
649
struct OplogRecycleStats {
650
    // Total oplog count scanned per round
651
    std::atomic<int64_t> total_num {0};
652
    // Oplogs not recycled this round (per round, written to mBvarStatus)
653
    std::atomic<int64_t> not_recycled_num {0};
654
    // Recycle failures (per round, accumulated to mBvarIntAdder at end)
655
    std::atomic<int64_t> failed_num {0};
656
    // Per-oplog-type recycled counts (incremented after successful commit)
657
    std::atomic<int64_t> recycled_commit_partition {0};
658
    std::atomic<int64_t> recycled_drop_partition {0};
659
    std::atomic<int64_t> recycled_commit_index {0};
660
    std::atomic<int64_t> recycled_drop_index {0};
661
    std::atomic<int64_t> recycled_update_tablet {0};
662
    std::atomic<int64_t> recycled_compaction {0};
663
    std::atomic<int64_t> recycled_schema_change {0};
664
    std::atomic<int64_t> recycled_commit_txn {0};
665
};
666
667
// Helper class to check if operation logs can be recycled based on snapshots and versionstamps
668
class OperationLogRecycleChecker {
669
public:
670
    OperationLogRecycleChecker(std::string_view instance_id, TxnKv* txn_kv,
671
                               const InstanceInfoPB& instance_info)
672
35
            : instance_id_(instance_id), txn_kv_(txn_kv), instance_info_(instance_info) {}
673
674
    // Initialize the checker by loading snapshots and setting max version stamp
675
    int init();
676
677
    // Check if an operation log can be recycled
678
    bool can_recycle(const Versionstamp& log_versionstamp, int64_t log_min_timestamp,
679
                     OperationLogReferenceInfo* reference_info) const;
680
681
0
    Versionstamp max_versionstamp() const { return max_versionstamp_; }
682
683
28
    const std::vector<std::pair<SnapshotPB, Versionstamp>>& get_snapshots() const {
684
28
        return snapshots_;
685
28
    }
686
687
private:
688
    std::string_view instance_id_;
689
    TxnKv* txn_kv_;
690
    const InstanceInfoPB& instance_info_;
691
    Versionstamp max_versionstamp_;
692
    Versionstamp source_snapshot_versionstamp_;
693
    std::map<Versionstamp, size_t> snapshot_indexes_;
694
    std::vector<std::pair<SnapshotPB, Versionstamp>> snapshots_;
695
};
696
697
class SnapshotDataSizeCalculator {
698
public:
699
    SnapshotDataSizeCalculator(std::string_view instance_id, std::shared_ptr<TxnKv> txn_kv)
700
29
            : instance_id_(instance_id), txn_kv_(std::move(txn_kv)) {}
701
702
    void init(const std::vector<std::pair<SnapshotPB, Versionstamp>>& snapshots);
703
704
    int calculate_operation_log_data_size(const std::string_view& log_key,
705
                                          OperationLogPB& operation_log,
706
                                          OperationLogReferenceInfo& reference_info);
707
708
    int save_snapshot_data_size_with_retry();
709
710
private:
711
    int get_all_index_partitions(int64_t db_id, int64_t table_id, int64_t index_id,
712
                                 std::vector<int64_t>* partition_ids);
713
    int get_index_partition_data_size(int64_t db_id, int64_t table_id, int64_t index_id,
714
                                      int64_t partition_id, int64_t* data_size);
715
    int save_operation_log(const std::string_view& log_key, OperationLogPB& operation_log);
716
    int save_snapshot_data_size();
717
718
    std::string_view instance_id_;
719
    std::shared_ptr<TxnKv> txn_kv_;
720
721
    int64_t instance_retained_data_size_ = 0;
722
    std::map<Versionstamp, int64_t> retained_data_size_;
723
    std::set<std::string> calculated_partitions_;
724
};
725
726
} // namespace doris::cloud