/src/rocksdb/utilities/transactions/write_prepared_txn.h
Line | Count | Source |
1 | | // Copyright (c) 2011-present, Facebook, Inc. All rights reserved. |
2 | | // This source code is licensed under both the GPLv2 (found in the |
3 | | // COPYING file in the root directory) and Apache 2.0 License |
4 | | // (found in the LICENSE.Apache file in the root directory). |
5 | | |
6 | | #pragma once |
7 | | |
8 | | #include <algorithm> |
9 | | #include <atomic> |
10 | | #include <mutex> |
11 | | #include <stack> |
12 | | #include <string> |
13 | | #include <unordered_map> |
14 | | #include <vector> |
15 | | |
16 | | #include "db/write_callback.h" |
17 | | #include "rocksdb/db.h" |
18 | | #include "rocksdb/slice.h" |
19 | | #include "rocksdb/snapshot.h" |
20 | | #include "rocksdb/status.h" |
21 | | #include "rocksdb/types.h" |
22 | | #include "rocksdb/utilities/transaction.h" |
23 | | #include "rocksdb/utilities/transaction_db.h" |
24 | | #include "rocksdb/utilities/write_batch_with_index.h" |
25 | | #include "util/autovector.h" |
26 | | #include "utilities/transactions/pessimistic_transaction.h" |
27 | | #include "utilities/transactions/pessimistic_transaction_db.h" |
28 | | #include "utilities/transactions/transaction_base.h" |
29 | | #include "utilities/transactions/transaction_util.h" |
30 | | |
31 | | namespace ROCKSDB_NAMESPACE { |
32 | | |
33 | | class WritePreparedTxnDB; |
34 | | |
35 | | // This impl could write to DB also uncommitted data and then later tell apart |
36 | | // committed data from uncommitted data. Uncommitted data could be after the |
37 | | // Prepare phase in 2PC (WritePreparedTxn) or before that |
38 | | // (WriteUnpreparedTxnImpl). |
39 | | // |
40 | | // == Concrete example: WritePrepared 2PC transaction == |
41 | | // |
42 | | // User code: |
43 | | // |
44 | | // Transaction* txn = db->BeginTransaction(write_opts, txn_opts); |
45 | | // txn->SetName("txn1"); |
46 | | // txn->Put("key1", "value1"); // buffered in WriteBatch, nothing written |
47 | | // yet txn->Prepare(); // Phase 1 txn->Commit(); // Phase 2 |
48 | | // |
49 | | // -- Phase 1: Prepare (PrepareInternal) -- |
50 | | // |
51 | | // The Prepare call (write_prepared_txn.cc PrepareInternal) calls: |
52 | | // |
53 | | // db_impl_->WriteImpl(write_options, GetWriteBatch(), |
54 | | // ..., !DISABLE_MEMTABLE, ...); |
55 | | // |
56 | | // !DISABLE_MEMTABLE is false -- memtable is enabled. This is the defining |
57 | | // characteristic of "WritePrepared": the actual data (Put("key1", "value1")) |
58 | | // is written to the memtable at Prepare time. |
59 | | // |
60 | | // Because disable_memtable == false, the routing check at |
61 | | // db_impl_write.cc:502 is not taken. The write goes through the main write |
62 | | // queue (write_thread_), which handles both WAL and memtable: |
63 | | // |
64 | | // Destination | What gets written | Sequence |
65 | | // ------------|--------------------------------------------|----------- |
66 | | // WAL | Put(key1, value1) + EndPrepare(txn1) | prepare_seq |
67 | | // Memtable | Put(key1, value1) | prepare_seq |
68 | | // |
69 | | // The data is now durable (WAL) and in the memtable, but not yet visible |
70 | | // to readers. Readers use GetLastPublishedSequence() which consults a |
71 | | // commit map -- since prepare_seq is in the PreparedHeap but not yet in the |
72 | | // CommitCache, readers know this data is uncommitted and skip it. |
73 | | // |
74 | | // -- Phase 2: Commit (CommitInternal) -- |
75 | | // |
76 | | // The Commit call (write_prepared_txn.cc CommitInternal) calls: |
77 | | // |
78 | | // db_impl_->WriteImpl(write_options_, working_batch, |
79 | | // ..., disable_memtable, ...); |
80 | | // |
81 | | // In the typical case (do_one_write == true, i.e., the commit-time batch |
82 | | // is empty or has no data), disable_memtable is true. Now the routing |
83 | | // check at db_impl_write.cc:502 is taken: |
84 | | // |
85 | | // if (two_write_queues_ && disable_memtable) { |
86 | | // return WriteImplWALOnly(&nonmem_write_thread_, ...); |
87 | | // } |
88 | | // |
89 | | // The commit goes through the second write queue (nonmem_write_thread_), |
90 | | // WAL only: |
91 | | // |
92 | | // Destination | What gets written | Sequence |
93 | | // ------------|---------------------|----------- |
94 | | // WAL | Commit(txn1) marker | commit_seq |
95 | | // Memtable | Nothing | -- |
96 | | // |
97 | | // The PreReleaseCallback (WritePreparedCommitEntryPreReleaseCallback) |
98 | | // updates the CommitCache to record that prepare_seq was committed at |
99 | | // commit_seq. After this, readers consulting the commit map will see that |
100 | | // the data at prepare_seq is committed and therefore visible. |
101 | | // |
102 | | // -- Why two queues help -- |
103 | | // |
104 | | // The Commit phase doesn't touch the memtable -- it only writes a small |
105 | | // marker to WAL and updates an in-memory commit map. By routing this |
106 | | // through a separate queue, Commit writes don't have to wait behind other |
107 | | // transactions' Prepare writes (which do the expensive memtable insertion |
108 | | // on the main queue). This is the optimization mentioned in the options |
109 | | // comment about MySQL 2PC where commits are serial. |
110 | | // |
111 | | // -- Sequence number flow -- |
112 | | // |
113 | | // last_sequence_ | last_allocated_seq | |
114 | | // last_published_seq |
115 | | // ---------------|--------------------|------------------- |
116 | | // Before Prepare: 9 | 9 | 9 |
117 | | // |
118 | | // Prepare (main queue): |
119 | | // FetchAdd alloc seq 9 | 10 | 9 |
120 | | // Write WAL + memtable |
121 | | // SetLastSequence 10 | 10 | 9 |
122 | | // (published_seq not advanced yet -- data is uncommitted) |
123 | | // |
124 | | // Commit (2nd queue): |
125 | | // FetchAdd alloc seq 10 | 11 | 9 |
126 | | // Write WAL only |
127 | | // Update CommitCache |
128 | | // SetLastPublishedSeq 10 | 11 | 11 |
129 | | // |
130 | | class WritePreparedTxn : public PessimisticTransaction { |
131 | | public: |
132 | | WritePreparedTxn(WritePreparedTxnDB* db, const WriteOptions& write_options, |
133 | | const TransactionOptions& txn_options); |
134 | | // No copying allowed |
135 | | WritePreparedTxn(const WritePreparedTxn&) = delete; |
136 | | void operator=(const WritePreparedTxn&) = delete; |
137 | | |
138 | 0 | virtual ~WritePreparedTxn() {} |
139 | | |
140 | | // To make WAL commit markers visible, the snapshot will be based on the last |
141 | | // seq in the WAL that is also published, LastPublishedSequence, as opposed to |
142 | | // the last seq in the memtable. |
143 | | using Transaction::Get; |
144 | | Status Get(const ReadOptions& _read_options, |
145 | | ColumnFamilyHandle* column_family, const Slice& key, |
146 | | PinnableSlice* value) override; |
147 | | |
148 | | using Transaction::MultiGet; |
149 | | void MultiGet(const ReadOptions& _read_options, |
150 | | ColumnFamilyHandle* column_family, const size_t num_keys, |
151 | | const Slice* keys, PinnableSlice* values, Status* statuses, |
152 | | const bool sorted_input = false) override; |
153 | | |
154 | | // Note: The behavior is undefined in presence of interleaved writes to the |
155 | | // same transaction. |
156 | | // To make WAL commit markers visible, the snapshot will be |
157 | | // based on the last seq in the WAL that is also published, |
158 | | // LastPublishedSequence, as opposed to the last seq in the memtable. |
159 | | using Transaction::GetIterator; |
160 | | Iterator* GetIterator(const ReadOptions& options) override; |
161 | | Iterator* GetIterator(const ReadOptions& options, |
162 | | ColumnFamilyHandle* column_family) override; |
163 | | |
164 | | std::unique_ptr<Iterator> GetCoalescingIterator( |
165 | | const ReadOptions& read_options, |
166 | | const std::vector<ColumnFamilyHandle*>& column_families) override; |
167 | | |
168 | | std::unique_ptr<AttributeGroupIterator> GetAttributeGroupIterator( |
169 | | const ReadOptions& read_options, |
170 | | const std::vector<ColumnFamilyHandle*>& column_families) override; |
171 | | |
172 | | void SetSnapshot() override; |
173 | | |
174 | | protected: |
175 | | void Initialize(const TransactionOptions& txn_options) override; |
176 | | // Override the protected SetId to make it visible to the friend class |
177 | | // WritePreparedTxnDB |
178 | 0 | inline void SetId(uint64_t id) override { Transaction::SetId(id); } |
179 | | |
180 | | private: |
181 | | friend class WritePreparedTransactionTest_BasicRecoveryTest_Test; |
182 | | friend class WritePreparedTxnDB; |
183 | | friend class WriteUnpreparedTxnDB; |
184 | | friend class WriteUnpreparedTxn; |
185 | | |
186 | | using Transaction::GetImpl; |
187 | | Status GetImpl(const ReadOptions& options, ColumnFamilyHandle* column_family, |
188 | | const Slice& key, PinnableSlice* value) override; |
189 | | |
190 | | Status PrepareInternal() override; |
191 | | |
192 | | Status CommitWithoutPrepareInternal() override; |
193 | | |
194 | | Status CommitBatchInternal(WriteBatch* batch, size_t batch_cnt) override; |
195 | | |
196 | | // Since the data is already written to memtables at the Prepare phase, the |
197 | | // commit entails writing only a commit marker in the WAL. The sequence number |
198 | | // of the commit marker is then the commit timestamp of the transaction. To |
199 | | // make WAL commit markers visible, the snapshot will be based on the last seq |
200 | | // in the WAL that is also published, LastPublishedSequence, as opposed to the |
201 | | // last seq in the memtable. |
202 | | Status CommitInternal() override; |
203 | | |
204 | | Status RollbackInternal() override; |
205 | | |
206 | | Status ValidateSnapshot(ColumnFamilyHandle* column_family, const Slice& key, |
207 | | SequenceNumber* tracked_at_seq) override; |
208 | | |
209 | | Status RebuildFromWriteBatch(WriteBatch* src_batch) override; |
210 | | |
211 | | WritePreparedTxnDB* wpt_db_; |
212 | | // Number of sub-batches in prepare |
213 | | size_t prepare_batch_cnt_ = 0; |
214 | | }; |
215 | | |
216 | | } // namespace ROCKSDB_NAMESPACE |