Coverage Report

Created: 2026-09-28 07:52

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/rocksdb/db/flush_job.h
Line
Count
Source
1
//  Copyright (c) 2011-present, Facebook, Inc.  All rights reserved.
2
//  This source code is licensed under both the GPLv2 (found in the
3
//  COPYING file in the root directory) and Apache 2.0 License
4
//  (found in the LICENSE.Apache file in the root directory).
5
//
6
// Copyright (c) 2011 The LevelDB Authors. All rights reserved.
7
// Use of this source code is governed by a BSD-style license that can be
8
// found in the LICENSE file. See the AUTHORS file for names of contributors.
9
#pragma once
10
11
#include <atomic>
12
#include <deque>
13
#include <limits>
14
#include <list>
15
#include <set>
16
#include <string>
17
#include <utility>
18
#include <vector>
19
20
#include "db/blob/blob_file_completion_callback.h"
21
#include "db/column_family.h"
22
#include "db/flush_scheduler.h"
23
#include "db/internal_stats.h"
24
#include "db/job_context.h"
25
#include "db/log_writer.h"
26
#include "db/logs_with_prep_tracker.h"
27
#include "db/memtable_list.h"
28
#include "db/seqno_to_time_mapping.h"
29
#include "db/snapshot_impl.h"
30
#include "db/version_edit.h"
31
#include "db/write_controller.h"
32
#include "db/write_thread.h"
33
#include "logging/event_logger.h"
34
#include "monitoring/instrumented_mutex.h"
35
#include "options/db_options.h"
36
#include "port/port.h"
37
#include "rocksdb/db.h"
38
#include "rocksdb/env.h"
39
#include "rocksdb/listener.h"
40
#include "rocksdb/memtablerep.h"
41
#include "rocksdb/wal_iterator.h"
42
#include "util/autovector.h"
43
#include "util/stop_watch.h"
44
#include "util/thread_local.h"
45
46
namespace ROCKSDB_NAMESPACE {
47
48
class DBImpl;
49
class MemTable;
50
class SnapshotChecker;
51
class TableCache;
52
class Version;
53
class VersionEdit;
54
class VersionSet;
55
class Arena;
56
57
class FlushJob {
58
 public:
59
  // TODO(icanadi) make effort to reduce number of parameters here
60
  // IMPORTANT: mutable_cf_options needs to be alive while FlushJob is alive
61
  FlushJob(const std::string& dbname, ColumnFamilyData* cfd,
62
           const ImmutableDBOptions& db_options,
63
           const MutableCFOptions& mutable_cf_options, uint64_t max_memtable_id,
64
           const FileOptions& file_options, VersionSet* versions,
65
           InstrumentedMutex* db_mutex, std::atomic<bool>* shutting_down,
66
           JobContext* job_context, FlushReason flush_reason,
67
           LogBuffer* log_buffer, FSDirectory* db_directory,
68
           FSDirectory* output_file_directory,
69
           CompressionType output_compression, Statistics* stats,
70
           EventLogger* event_logger, bool measure_io_stats,
71
           const bool sync_output_directory, const bool write_manifest,
72
           Env::Priority thread_pri, const std::shared_ptr<IOTracer>& io_tracer,
73
           std::shared_ptr<const SeqnoToTimeMapping> seqno_to_time_mapping,
74
           const std::string& db_id = "", const std::string& db_session_id = "",
75
           std::string full_history_ts_low = "",
76
           BlobFileCompletionCallback* blob_callback = nullptr,
77
           bool fast_sst_open = false);
78
79
  ~FlushJob();
80
81
  // Require db_mutex held.
82
  // Once PickMemTable() is called, either Run() or Cancel() has to be called.
83
  void PickMemTable();
84
  // @param skip_since_bg_error If not nullptr and if atomic_flush=false,
85
  // then it is set to true if flush installation is skipped and memtable
86
  // is rolled back due to existing background error.
87
  Status Run(LogsWithPrepTracker* prep_tracker = nullptr,
88
             FileMetaData* file_meta = nullptr,
89
             bool* switched_to_mempurge = nullptr,
90
             bool* skipped_since_bg_error = nullptr,
91
             ErrorHandler* error_handler = nullptr);
92
  void Cancel();
93
0
  const autovector<ReadOnlyMemTable*>& GetMemTables() const { return mems_; }
94
95
  // Returns the log number recorded in the flush VersionEdit after
96
  // PickMemTable() initializes `edit_`.
97
0
  uint64_t GetLogNumber() const {
98
0
    assert(edit_ != nullptr);
99
0
    return edit_->GetLogNumber();
100
0
  }
101
102
  // Stashes write-path blob files so WriteLevel0Table() can add them to the
103
  // same VersionEdit as the flushed SST.
104
0
  void AddExternalBlobFileAdditions(std::vector<BlobFileAddition>&& additions) {
105
0
    external_blob_file_additions_ = std::move(additions);
106
0
  }
107
108
  // Stashes write-path initial-garbage updates so they are committed with the
109
  // same VersionEdit as the matching blob-file additions and flushed SST.
110
0
  void AddExternalBlobFileGarbages(std::vector<BlobFileGarbage>&& garbages) {
111
0
    external_blob_file_garbages_ = std::move(garbages);
112
0
  }
113
114
  // Transfers back any prepared blob file additions that were not consumed by
115
  // the flush.
116
0
  std::vector<BlobFileAddition> TakeExternalBlobFileAdditions() {
117
0
    return std::move(external_blob_file_additions_);
118
0
  }
119
120
  // Transfers back any prepared blob-file garbage updates that were not
121
  // consumed by the flush.
122
0
  std::vector<BlobFileGarbage> TakeExternalBlobFileGarbages() {
123
0
    return std::move(external_blob_file_garbages_);
124
0
  }
125
126
2.33k
  std::list<std::unique_ptr<FlushJobInfo>>* GetCommittedFlushJobsInfo() {
127
2.33k
    return &committed_flush_jobs_info_;
128
2.33k
  }
129
130
 private:
131
  friend class FlushJobTest_GetRateLimiterPriorityForWrite_Test;
132
133
  void ReportStartedFlush();
134
  static void ReportFlushInputSize(const autovector<ReadOnlyMemTable*>& mems);
135
  void RecordFlushIOStats();
136
  Status WriteLevel0Table();
137
138
  // Memtable Garbage Collection algorithm: a MemPurge takes the list
139
  // of immutable memtables and filters out (or "purge") the outdated bytes
140
  // out of it. The output (the filtered bytes, or "useful payload") is
141
  // then transfered into a new memtable. If this memtable is filled, then
142
  // the mempurge is aborted and rerouted to a regular flush process. Else,
143
  // depending on the heuristics, placed onto the immutable memtable list.
144
  // The addition to the imm list will not trigger a flush operation. The
145
  // flush of the imm list will instead be triggered once the mutable memtable
146
  // is added to the imm list.
147
  // This process is typically intended for workloads with heavy overwrites
148
  // when we want to avoid SSD writes (and reads) as much as possible.
149
  // "MemPurge" is an experimental feature still at a very early stage
150
  // of development. At the moment it is only compatible with the Get, Put,
151
  // Delete operations as well as Iterators and CompactionFilters.
152
  // For this early version, "MemPurge" is called by setting the
153
  // options.experimental_mempurge_threshold value as >0.0. When this is
154
  // the case, ALL automatic flush operations (kWRiteBufferManagerFull) will
155
  // first go through the MemPurge process. Therefore, we strongly
156
  // recommend all users not to set this flag as true given that the MemPurge
157
  // process has not matured yet.
158
  Status MemPurge();
159
  bool MemPurgeDecider(double threshold);
160
  // The rate limiter priority (io_priority) is determined dynamically here.
161
  Env::IOPriority GetRateLimiterPriority();
162
  std::unique_ptr<FlushJobInfo> GetFlushJobInfo() const;
163
164
  // Require db_mutex held.
165
  // Called only when UDT feature is enabled and
166
  // `persist_user_defined_timestamps` flag is false. Because we will refrain
167
  // from flushing as long as there are still UDTs in a memtable that hasn't
168
  // expired w.r.t `full_history_ts_low`. However, flush is continued if there
169
  // is risk of entering write stall mode. In that case, we need
170
  // to track the effective cutoff timestamp below which all the udts are
171
  // removed because of flush, and use it to increase `full_history_ts_low` if
172
  // the effective cutoff timestamp is newer. See
173
  // `MaybeIncreaseFullHistoryTsLowToAboveCutoffUDT` for details.
174
  void GetEffectiveCutoffUDTForPickedMemTables();
175
176
  // If this column family enables tiering feature, it will find the current
177
  // `preclude_last_level_min_seqno_`, and the smaller one between this and
178
  // the `earliset_snapshot_` will later be announced to user property
179
  // collectors. It indicates to tiering use cases which data are old enough to
180
  // be placed on the last level.
181
  void GetPrecludeLastLevelMinSeqno();
182
183
  Status MaybeIncreaseFullHistoryTsLowToAboveCutoffUDT();
184
185
  const std::string& dbname_;
186
  const std::string db_id_;
187
  const std::string db_session_id_;
188
  ColumnFamilyData* cfd_;
189
  const ImmutableDBOptions& db_options_;
190
  const MutableCFOptions& mutable_cf_options_;
191
  // A variable storing the largest memtable id to flush in this
192
  // flush job. RocksDB uses this variable to select the memtables to flush in
193
  // this job. All memtables in this column family with an ID smaller than or
194
  // equal to max_memtable_id_ will be selected for flush.
195
  uint64_t max_memtable_id_;
196
  FileOptions file_options_;
197
  VersionSet* versions_;
198
  InstrumentedMutex* db_mutex_;
199
  std::atomic<bool>* shutting_down_;
200
  SequenceNumber earliest_snapshot_;
201
  JobContext* job_context_;
202
  FlushReason flush_reason_;
203
  LogBuffer* log_buffer_;
204
  FSDirectory* db_directory_;
205
  FSDirectory* output_file_directory_;
206
  CompressionType output_compression_;
207
  Statistics* stats_;
208
  EventLogger* event_logger_;
209
  TableProperties table_properties_;
210
  bool measure_io_stats_;
211
  // True if this flush job should call fsync on the output directory. False
212
  // otherwise.
213
  // Usually sync_output_directory_ is true. A flush job needs to call sync on
214
  // the output directory before committing to the MANIFEST.
215
  // However, an individual flush job does not have to call sync on the output
216
  // directory if it is part of an atomic flush. After all flush jobs in the
217
  // atomic flush succeed, call sync once on each distinct output directory.
218
  const bool sync_output_directory_;
219
  // True if this flush job should write to MANIFEST after successfully
220
  // flushing memtables. False otherwise.
221
  // Usually write_manifest_ is true. A flush job commits to the MANIFEST after
222
  // flushing the memtables.
223
  // However, an individual flush job cannot rashly write to the MANIFEST
224
  // immediately after it finishes the flush if it is part of an atomic flush.
225
  // In this case, only after all flush jobs succeed in flush can RocksDB
226
  // commit to the MANIFEST.
227
  const bool write_manifest_;
228
  // The current flush job can commit flush result of a concurrent flush job.
229
  // We collect FlushJobInfo of all jobs committed by current job and fire
230
  // OnFlushCompleted for them.
231
  std::list<std::unique_ptr<FlushJobInfo>> committed_flush_jobs_info_;
232
233
  // Variables below are set by PickMemTable():
234
  FileMetaData meta_;
235
  // Memtables to be flushed by this job.
236
  // Ordered by increasing memtable id, i.e., oldest memtable first.
237
  autovector<ReadOnlyMemTable*> mems_;
238
  VersionEdit* edit_;
239
  Version* base_;
240
  bool pick_memtable_called;
241
  Env::Priority thread_pri_;
242
243
  const std::shared_ptr<IOTracer> io_tracer_;
244
  SystemClock* clock_;
245
246
  const std::string full_history_ts_low_;
247
  BlobFileCompletionCallback* blob_callback_;
248
  bool fast_sst_open_;
249
  // Write-path blob files that should be committed with this flush.
250
  std::vector<BlobFileAddition> external_blob_file_additions_;
251
  // Initial garbage for write-path blob files that were partially abandoned
252
  // before their owning flush committed.
253
  std::vector<BlobFileGarbage> external_blob_file_garbages_;
254
255
  // Shared copy of DB's seqno to time mapping stored in SuperVersion. The
256
  // ownership is shared with this FlushJob when it's created.
257
  // FlushJob accesses and ref counts immutable MemTables directly via
258
  // `MemTableListVersion` instead of ref `SuperVersion`, so we need to give
259
  // the flush job shared ownership of the mapping.
260
  // Note this is only installed when seqno to time recording feature is
261
  // enables, so it could be nullptr.
262
  std::shared_ptr<const SeqnoToTimeMapping> seqno_to_time_mapping_;
263
264
  // Keeps track of the newest user-defined timestamp for this flush job if
265
  // `persist_user_defined_timestamps` flag is false.
266
  std::string cutoff_udt_;
267
268
  // The current minimum seqno that compaction jobs will preclude the data from
269
  // the last level. Data with seqnos larger than this or larger than
270
  // `earliest_snapshot_` will be output to the proximal level had it gone
271
  // through a compaction to the last level.
272
  SequenceNumber preclude_last_level_min_seqno_ = kMaxSequenceNumber;
273
};
274
275
}  // namespace ROCKSDB_NAMESPACE