Coverage Report

Created: 2026-09-28 06:55

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/postgres/src/backend/utils/time/snapmgr.c
Line
Count
Source
1
/*-------------------------------------------------------------------------
2
 *
3
 * snapmgr.c
4
 *    PostgreSQL snapshot manager
5
 *
6
 * The following functions return an MVCC snapshot that can be used in tuple
7
 * visibility checks:
8
 *
9
 * - GetTransactionSnapshot
10
 * - GetLatestSnapshot
11
 * - GetCatalogSnapshot
12
 * - GetNonHistoricCatalogSnapshot
13
 *
14
 * Each of these functions returns a reference to a statically allocated
15
 * snapshot.  The statically allocated snapshot is subject to change on any
16
 * snapshot-related function call, and should not be used directly.  Instead,
17
 * call PushActiveSnapshot() or RegisterSnapshot() to create a longer-lived
18
 * copy and use that.
19
 *
20
 * We keep track of snapshots in two ways: those "registered" by resowner.c,
21
 * and the "active snapshot" stack.  All snapshots in either of them live in
22
 * persistent memory.  When a snapshot is no longer in any of these lists
23
 * (tracked by separate refcounts on each snapshot), its memory can be freed.
24
 *
25
 * In addition to the above-mentioned MVCC snapshots, there are some special
26
 * snapshots like SnapshotSelf, SnapshotAny, and "dirty" snapshots.  They can
27
 * only be used in limited contexts and cannot be registered or pushed to the
28
 * active stack.
29
 *
30
 * ActiveSnapshot stack
31
 * --------------------
32
 *
33
 * Most visibility checks use the current "active snapshot" returned by
34
 * GetActiveSnapshot().  When running normal queries, the active snapshot is
35
 * set when query execution begins based on the transaction isolation level.
36
 *
37
 * The active snapshot is tracked in a stack so that the currently active one
38
 * is at the top of the stack.  It mirrors the process call stack: whenever we
39
 * recurse or switch context to fetch rows from a different portal for
40
 * example, the appropriate snapshot is pushed to become the active snapshot,
41
 * and popped on return.  Once upon a time, ActiveSnapshot was just a global
42
 * variable that was saved and restored similar to CurrentMemoryContext, but
43
 * nowadays it's managed as a separate data structure so that we can keep
44
 * track of which snapshots are in use and reset MyProc->xmin when there is no
45
 * active snapshot.
46
 *
47
 * However, there are a couple of exceptions where the active snapshot stack
48
 * does not strictly mirror the call stack:
49
 *
50
 * - VACUUM and a few other utility commands manage their own transactions,
51
 *   which take their own snapshots.  They are called with an active snapshot
52
 *   set, like most utility commands, but they pop the active snapshot that
53
 *   was pushed by the caller.  PortalRunUtility knows about the possibility
54
 *   that the snapshot it pushed is no longer active on return.
55
 *
56
 * - When COMMIT or ROLLBACK is executed within a procedure or DO-block, the
57
 *   active snapshot stack is destroyed, and re-established later when
58
 *   subsequent statements in the procedure are executed.  There are many
59
 *   limitations on when in-procedure COMMIT/ROLLBACK is allowed; one such
60
 *   limitation is that all the snapshots on the active snapshot stack are
61
 *   known to portals that are being executed, which makes it safe to reset
62
 *   the stack.  See EnsurePortalSnapshotExists().
63
 *
64
 * Registered snapshots
65
 * --------------------
66
 *
67
 * In addition to snapshots pushed to the active snapshot stack, a snapshot
68
 * can be registered with a resource owner.
69
 *
70
 * The FirstXactSnapshot, if any, is treated a bit specially: we increment its
71
 * regd_count and list it in RegisteredSnapshots, but this reference is not
72
 * tracked by a resource owner. We used to use the TopTransactionResourceOwner
73
 * to track this snapshot reference, but that introduces logical circularity
74
 * and thus makes it impossible to clean up in a sane fashion.  It's better to
75
 * handle this reference as an internally-tracked registration, so that this
76
 * module is entirely lower-level than ResourceOwners.
77
 *
78
 * Likewise, any snapshots that have been exported by pg_export_snapshot
79
 * have regd_count = 1 and are listed in RegisteredSnapshots, but are not
80
 * tracked by any resource owner.
81
 *
82
 * Likewise, the CatalogSnapshot is listed in RegisteredSnapshots when it
83
 * is valid, but is not tracked by any resource owner.
84
 *
85
 * The same is true for historic snapshots used during logical decoding,
86
 * their lifetime is managed separately (as they live longer than one xact.c
87
 * transaction).
88
 *
89
 * These arrangements let us reset MyProc->xmin when there are no snapshots
90
 * referenced by this transaction, and advance it when the one with oldest
91
 * Xmin is no longer referenced.  For simplicity however, only registered
92
 * snapshots not active snapshots participate in tracking which one is oldest;
93
 * we don't try to change MyProc->xmin except when the active-snapshot
94
 * stack is empty.
95
 *
96
 *
97
 * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
98
 * Portions Copyright (c) 1994, Regents of the University of California
99
 *
100
 * IDENTIFICATION
101
 *    src/backend/utils/time/snapmgr.c
102
 *
103
 *-------------------------------------------------------------------------
104
 */
105
#include "postgres.h"
106
107
#include <sys/stat.h>
108
#include <unistd.h>
109
110
#include "access/subtrans.h"
111
#include "access/transam.h"
112
#include "access/xact.h"
113
#include "datatype/timestamp.h"
114
#include "lib/pairingheap.h"
115
#include "miscadmin.h"
116
#include "port/pg_lfind.h"
117
#include "storage/fd.h"
118
#include "storage/predicate.h"
119
#include "storage/proc.h"
120
#include "storage/procarray.h"
121
#include "utils/builtins.h"
122
#include "utils/injection_point.h"
123
#include "utils/memutils.h"
124
#include "utils/resowner.h"
125
#include "utils/snapmgr.h"
126
#include "utils/syscache.h"
127
128
129
/*
130
 * CurrentSnapshot points to the only snapshot taken in transaction-snapshot
131
 * mode, and to the latest one taken in a read-committed transaction.
132
 * SecondarySnapshot is a snapshot that's always up-to-date as of the current
133
 * instant, even in transaction-snapshot mode.  It should only be used for
134
 * special-purpose code (say, RI checking.)  CatalogSnapshot points to an
135
 * MVCC snapshot intended to be used for catalog scans; we must invalidate it
136
 * whenever a system catalog change occurs.
137
 *
138
 * These SnapshotData structs are static to simplify memory allocation
139
 * (see the hack in GetSnapshotData to avoid repeated malloc/free).
140
 */
141
static SnapshotData CurrentSnapshotData = {SNAPSHOT_MVCC};
142
static SnapshotData SecondarySnapshotData = {SNAPSHOT_MVCC};
143
static SnapshotData CatalogSnapshotData = {SNAPSHOT_MVCC};
144
SnapshotData SnapshotSelfData = {SNAPSHOT_SELF};
145
SnapshotData SnapshotAnyData = {SNAPSHOT_ANY};
146
SnapshotData SnapshotToastData = {SNAPSHOT_TOAST};
147
148
/* Pointers to valid snapshots */
149
static Snapshot CurrentSnapshot = NULL;
150
static Snapshot SecondarySnapshot = NULL;
151
static Snapshot CatalogSnapshot = NULL;
152
static Snapshot HistoricSnapshot = NULL;
153
154
/*
155
 * These are updated by GetSnapshotData.  We initialize them this way
156
 * for the convenience of TransactionIdIsInProgress: even in bootstrap
157
 * mode, we don't want it to say that BootstrapTransactionId is in progress.
158
 */
159
TransactionId TransactionXmin = FirstNormalTransactionId;
160
TransactionId RecentXmin = FirstNormalTransactionId;
161
162
/* (table, ctid) => (cmin, cmax) mapping during timetravel */
163
static HTAB *tuplecid_data = NULL;
164
165
/*
166
 * Elements of the active snapshot stack.
167
 *
168
 * Each element here accounts for exactly one active_count on SnapshotData.
169
 *
170
 * NB: the code assumes that elements in this list are in non-increasing
171
 * order of as_level; also, the list must be NULL-terminated.
172
 */
173
typedef struct ActiveSnapshotElt
174
{
175
  Snapshot  as_snap;
176
  int     as_level;
177
  struct ActiveSnapshotElt *as_next;
178
} ActiveSnapshotElt;
179
180
/* Top of the stack of active snapshots */
181
static ActiveSnapshotElt *ActiveSnapshot = NULL;
182
183
/*
184
 * Currently registered Snapshots.  Ordered in a heap by xmin, so that we can
185
 * quickly find the one with lowest xmin, to advance our MyProc->xmin.
186
 */
187
static int  xmin_cmp(const pairingheap_node *a, const pairingheap_node *b,
188
           void *arg);
189
190
static pairingheap RegisteredSnapshots = {&xmin_cmp, NULL, NULL};
191
192
/* first GetTransactionSnapshot call in a transaction? */
193
bool    FirstSnapshotSet = false;
194
195
/*
196
 * Remember the serializable transaction snapshot, if any.  We cannot trust
197
 * FirstSnapshotSet in combination with IsolationUsesXactSnapshot(), because
198
 * GUC may be reset before us, changing the value of IsolationUsesXactSnapshot.
199
 */
200
static Snapshot FirstXactSnapshot = NULL;
201
202
/* Define pathname of exported-snapshot files */
203
0
#define SNAPSHOT_EXPORT_DIR "pg_snapshots"
204
205
/* Structure holding info about exported snapshot. */
206
typedef struct ExportedSnapshot
207
{
208
  char     *snapfile;
209
  Snapshot  snapshot;
210
} ExportedSnapshot;
211
212
/* Current xact's exported snapshots (a list of ExportedSnapshot structs) */
213
static List *exportedSnapshots = NIL;
214
215
/* Prototypes for local functions */
216
static Snapshot CopySnapshot(Snapshot snapshot);
217
static void UnregisterSnapshotNoOwner(Snapshot snapshot);
218
static void FreeSnapshot(Snapshot snapshot);
219
static void SnapshotResetXmin(void);
220
221
/* ResourceOwner callbacks to track snapshot references */
222
static void ResOwnerReleaseSnapshot(Datum res);
223
224
static const ResourceOwnerDesc snapshot_resowner_desc =
225
{
226
  .name = "snapshot reference",
227
  .release_phase = RESOURCE_RELEASE_AFTER_LOCKS,
228
  .release_priority = RELEASE_PRIO_SNAPSHOT_REFS,
229
  .ReleaseResource = ResOwnerReleaseSnapshot,
230
  .DebugPrint = NULL      /* the default message is fine */
231
};
232
233
/* Convenience wrappers over ResourceOwnerRemember/Forget */
234
static inline void
235
ResourceOwnerRememberSnapshot(ResourceOwner owner, Snapshot snap)
236
0
{
237
0
  ResourceOwnerRemember(owner, PointerGetDatum(snap), &snapshot_resowner_desc);
238
0
}
239
static inline void
240
ResourceOwnerForgetSnapshot(ResourceOwner owner, Snapshot snap)
241
0
{
242
0
  ResourceOwnerForget(owner, PointerGetDatum(snap), &snapshot_resowner_desc);
243
0
}
244
245
/*
246
 * Snapshot fields to be serialized.
247
 *
248
 * Only these fields need to be sent to the cooperating backend; the
249
 * remaining ones can (and must) be set by the receiver upon restore.
250
 */
251
typedef struct SerializedSnapshotData
252
{
253
  TransactionId xmin;
254
  TransactionId xmax;
255
  uint32    xcnt;
256
  int32   subxcnt;
257
  bool    suboverflowed;
258
  bool    takenDuringRecovery;
259
  CommandId curcid;
260
} SerializedSnapshotData;
261
262
/*
263
 * GetTransactionSnapshot
264
 *    Get the appropriate snapshot for a new query in a transaction.
265
 *
266
 * Note that the return value points at static storage that will be modified
267
 * by future calls and by CommandCounterIncrement().  Callers must call
268
 * RegisterSnapshot or PushActiveSnapshot on the returned snap before doing
269
 * any other non-trivial work that could invalidate it.
270
 */
271
Snapshot
272
GetTransactionSnapshot(void)
273
0
{
274
  /*
275
   * Return historic snapshot if doing logical decoding.
276
   *
277
   * Historic snapshots are only usable for catalog access, not for
278
   * general-purpose queries.  The caller is responsible for ensuring that
279
   * the snapshot is used correctly! (PostgreSQL code never calls this
280
   * during logical decoding, but extensions can do it.)
281
   */
282
0
  if (HistoricSnapshotActive())
283
0
  {
284
    /*
285
     * We'll never need a non-historic transaction snapshot in this
286
     * (sub-)transaction, so there's no need to be careful to set one up
287
     * for later calls to GetTransactionSnapshot().
288
     */
289
0
    Assert(!FirstSnapshotSet);
290
0
    return HistoricSnapshot;
291
0
  }
292
293
  /* First call in transaction? */
294
0
  if (!FirstSnapshotSet)
295
0
  {
296
    /*
297
     * Don't allow catalog snapshot to be older than xact snapshot.  Must
298
     * do this first to allow the empty-heap Assert to succeed.
299
     */
300
0
    InvalidateCatalogSnapshot();
301
302
0
    Assert(pairingheap_is_empty(&RegisteredSnapshots));
303
0
    Assert(FirstXactSnapshot == NULL);
304
305
0
    if (IsInParallelMode())
306
0
      elog(ERROR,
307
0
         "cannot take query snapshot during a parallel operation");
308
309
    /*
310
     * In transaction-snapshot mode, the first snapshot must live until
311
     * end of xact regardless of what the caller does with it, so we must
312
     * make a copy of it rather than returning CurrentSnapshotData
313
     * directly.  Furthermore, if we're running in serializable mode,
314
     * predicate.c needs to wrap the snapshot fetch in its own processing.
315
     */
316
0
    if (IsolationUsesXactSnapshot())
317
0
    {
318
      /* First, create the snapshot in CurrentSnapshotData */
319
0
      if (IsolationIsSerializable())
320
0
        CurrentSnapshot = GetSerializableTransactionSnapshot(&CurrentSnapshotData);
321
0
      else
322
0
        CurrentSnapshot = GetSnapshotData(&CurrentSnapshotData);
323
      /* Make a saved copy */
324
0
      CurrentSnapshot = CopySnapshot(CurrentSnapshot);
325
0
      FirstXactSnapshot = CurrentSnapshot;
326
      /* Mark it as "registered" in FirstXactSnapshot */
327
0
      FirstXactSnapshot->regd_count++;
328
0
      pairingheap_add(&RegisteredSnapshots, &FirstXactSnapshot->ph_node);
329
0
    }
330
0
    else
331
0
      CurrentSnapshot = GetSnapshotData(&CurrentSnapshotData);
332
333
0
    FirstSnapshotSet = true;
334
0
    return CurrentSnapshot;
335
0
  }
336
337
0
  if (IsolationUsesXactSnapshot())
338
0
    return CurrentSnapshot;
339
340
  /* Don't allow catalog snapshot to be older than xact snapshot. */
341
0
  InvalidateCatalogSnapshot();
342
343
0
  CurrentSnapshot = GetSnapshotData(&CurrentSnapshotData);
344
345
0
  return CurrentSnapshot;
346
0
}
347
348
/*
349
 * GetLatestSnapshot
350
 *    Get a snapshot that is up-to-date as of the current instant,
351
 *    even if we are executing in transaction-snapshot mode.
352
 */
353
Snapshot
354
GetLatestSnapshot(void)
355
0
{
356
  /*
357
   * We might be able to relax this, but nothing that could otherwise work
358
   * needs it.
359
   */
360
0
  if (IsInParallelMode())
361
0
    elog(ERROR,
362
0
       "cannot update SecondarySnapshot during a parallel operation");
363
364
  /*
365
   * So far there are no cases requiring support for GetLatestSnapshot()
366
   * during logical decoding, but it wouldn't be hard to add if required.
367
   */
368
0
  Assert(!HistoricSnapshotActive());
369
370
  /* If first call in transaction, go ahead and set the xact snapshot */
371
0
  if (!FirstSnapshotSet)
372
0
    return GetTransactionSnapshot();
373
374
0
  SecondarySnapshot = GetSnapshotData(&SecondarySnapshotData);
375
376
0
  return SecondarySnapshot;
377
0
}
378
379
/*
380
 * GetCatalogSnapshot
381
 *    Get a snapshot that is sufficiently up-to-date for scan of the
382
 *    system catalog with the specified OID.
383
 */
384
Snapshot
385
GetCatalogSnapshot(Oid relid)
386
0
{
387
  /*
388
   * Return historic snapshot while we're doing logical decoding, so we can
389
   * see the appropriate state of the catalog.
390
   *
391
   * This is the primary reason for needing to reset the system caches after
392
   * finishing decoding.
393
   */
394
0
  if (HistoricSnapshotActive())
395
0
    return HistoricSnapshot;
396
397
0
  return GetNonHistoricCatalogSnapshot(relid);
398
0
}
399
400
/*
401
 * GetNonHistoricCatalogSnapshot
402
 *    Get a snapshot that is sufficiently up-to-date for scan of the system
403
 *    catalog with the specified OID, even while historic snapshots are set
404
 *    up.
405
 */
406
Snapshot
407
GetNonHistoricCatalogSnapshot(Oid relid)
408
0
{
409
  /*
410
   * If the caller is trying to scan a relation that has no syscache, no
411
   * catcache invalidations will be sent when it is updated.  For a few key
412
   * relations, snapshot invalidations are sent instead.  If we're trying to
413
   * scan a relation for which neither catcache nor snapshot invalidations
414
   * are sent, we must refresh the snapshot every time.
415
   */
416
0
  if (CatalogSnapshot &&
417
0
    !RelationInvalidatesSnapshotsOnly(relid) &&
418
0
    !RelationHasSysCache(relid))
419
0
    InvalidateCatalogSnapshot();
420
421
0
  if (CatalogSnapshot == NULL)
422
0
  {
423
    /* Get new snapshot. */
424
0
    CatalogSnapshot = GetSnapshotData(&CatalogSnapshotData);
425
426
    /*
427
     * Make sure the catalog snapshot will be accounted for in decisions
428
     * about advancing PGPROC->xmin.  We could apply RegisterSnapshot, but
429
     * that would result in making a physical copy, which is overkill; and
430
     * it would also create a dependency on some resource owner, which we
431
     * do not want for reasons explained at the head of this file. Instead
432
     * just shove the CatalogSnapshot into the pairing heap manually. This
433
     * has to be reversed in InvalidateCatalogSnapshot, of course.
434
     *
435
     * NB: it had better be impossible for this to throw error, since the
436
     * CatalogSnapshot pointer is already valid.
437
     */
438
0
    pairingheap_add(&RegisteredSnapshots, &CatalogSnapshot->ph_node);
439
0
  }
440
441
0
  return CatalogSnapshot;
442
0
}
443
444
/*
445
 * InvalidateCatalogSnapshot
446
 *    Mark the current catalog snapshot, if any, as invalid
447
 *
448
 * We could change this API to allow the caller to provide more fine-grained
449
 * invalidation details, so that a change to relation A wouldn't prevent us
450
 * from using our cached snapshot to scan relation B, but so far there's no
451
 * evidence that the CPU cycles we spent tracking such fine details would be
452
 * well-spent.
453
 */
454
void
455
InvalidateCatalogSnapshot(void)
456
0
{
457
0
  if (CatalogSnapshot)
458
0
  {
459
0
    pairingheap_remove(&RegisteredSnapshots, &CatalogSnapshot->ph_node);
460
0
    CatalogSnapshot = NULL;
461
0
    SnapshotResetXmin();
462
0
    INJECTION_POINT("invalidate-catalog-snapshot-end", NULL);
463
0
  }
464
0
}
465
466
/*
467
 * InvalidateCatalogSnapshotConditionally
468
 *    Drop catalog snapshot if it's the only one we have
469
 *
470
 * This is called when we are about to wait for client input, so we don't
471
 * want to continue holding the catalog snapshot if it might mean that the
472
 * global xmin horizon can't advance.  However, if there are other snapshots
473
 * still active or registered, the catalog snapshot isn't likely to be the
474
 * oldest one, so we might as well keep it.
475
 */
476
void
477
InvalidateCatalogSnapshotConditionally(void)
478
0
{
479
0
  if (CatalogSnapshot &&
480
0
    ActiveSnapshot == NULL &&
481
0
    pairingheap_is_singular(&RegisteredSnapshots))
482
0
    InvalidateCatalogSnapshot();
483
0
}
484
485
/*
486
 * SnapshotSetCommandId
487
 *    Propagate CommandCounterIncrement into the static snapshots, if set
488
 */
489
void
490
SnapshotSetCommandId(CommandId curcid)
491
0
{
492
0
  if (!FirstSnapshotSet)
493
0
    return;
494
495
0
  if (CurrentSnapshot)
496
0
    CurrentSnapshot->curcid = curcid;
497
0
  if (SecondarySnapshot)
498
0
    SecondarySnapshot->curcid = curcid;
499
  /* Should we do the same with CatalogSnapshot? */
500
0
}
501
502
/*
503
 * SetTransactionSnapshot
504
 *    Set the transaction's snapshot from an imported MVCC snapshot.
505
 *
506
 * Note that this is very closely tied to GetTransactionSnapshot --- it
507
 * must take care of all the same considerations as the first-snapshot case
508
 * in GetTransactionSnapshot.
509
 */
510
static void
511
SetTransactionSnapshot(Snapshot sourcesnap, VirtualTransactionId *sourcevxid,
512
             int sourcepid, PGPROC *sourceproc)
513
0
{
514
  /* Caller should have checked this already */
515
0
  Assert(!FirstSnapshotSet);
516
517
  /* Better do this to ensure following Assert succeeds. */
518
0
  InvalidateCatalogSnapshot();
519
520
0
  Assert(pairingheap_is_empty(&RegisteredSnapshots));
521
0
  Assert(FirstXactSnapshot == NULL);
522
0
  Assert(!HistoricSnapshotActive());
523
524
  /*
525
   * Even though we are not going to use the snapshot it computes, we must
526
   * call GetSnapshotData, for two reasons: (1) to be sure that
527
   * CurrentSnapshotData's XID arrays have been allocated, and (2) to update
528
   * the state for GlobalVis*.
529
   */
530
0
  CurrentSnapshot = GetSnapshotData(&CurrentSnapshotData);
531
532
  /*
533
   * Now copy appropriate fields from the source snapshot.
534
   */
535
0
  CurrentSnapshot->xmin = sourcesnap->xmin;
536
0
  CurrentSnapshot->xmax = sourcesnap->xmax;
537
0
  CurrentSnapshot->xcnt = sourcesnap->xcnt;
538
0
  Assert(sourcesnap->xcnt <= GetMaxSnapshotXidCount());
539
0
  if (sourcesnap->xcnt > 0)
540
0
    memcpy(CurrentSnapshot->xip, sourcesnap->xip,
541
0
         sourcesnap->xcnt * sizeof(TransactionId));
542
0
  CurrentSnapshot->subxcnt = sourcesnap->subxcnt;
543
0
  Assert(sourcesnap->subxcnt <= GetMaxSnapshotSubxidCount());
544
0
  if (sourcesnap->subxcnt > 0)
545
0
    memcpy(CurrentSnapshot->subxip, sourcesnap->subxip,
546
0
         sourcesnap->subxcnt * sizeof(TransactionId));
547
0
  CurrentSnapshot->suboverflowed = sourcesnap->suboverflowed;
548
0
  CurrentSnapshot->takenDuringRecovery = sourcesnap->takenDuringRecovery;
549
  /* NB: curcid should NOT be copied, it's a local matter */
550
551
0
  CurrentSnapshot->snapXactCompletionCount = 0;
552
553
  /*
554
   * Now we have to fix what GetSnapshotData did with MyProc->xmin and
555
   * TransactionXmin.  There is a race condition: to make sure we are not
556
   * causing the global xmin to go backwards, we have to test that the
557
   * source transaction is still running, and that has to be done
558
   * atomically. So let procarray.c do it.
559
   *
560
   * Note: in serializable mode, predicate.c will do this a second time. It
561
   * doesn't seem worth contorting the logic here to avoid two calls,
562
   * especially since it's not clear that predicate.c *must* do this.
563
   */
564
0
  if (sourceproc != NULL)
565
0
  {
566
0
    if (!ProcArrayInstallRestoredXmin(CurrentSnapshot->xmin, sourceproc))
567
0
      ereport(ERROR,
568
0
          (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE),
569
0
           errmsg("could not import the requested snapshot"),
570
0
           errdetail("The source transaction is not running anymore.")));
571
0
  }
572
0
  else if (!ProcArrayInstallImportedXmin(CurrentSnapshot->xmin, sourcevxid))
573
0
    ereport(ERROR,
574
0
        (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE),
575
0
         errmsg("could not import the requested snapshot"),
576
0
         errdetail("The source process with PID %d is not running anymore.",
577
0
               sourcepid)));
578
579
  /*
580
   * In transaction-snapshot mode, the first snapshot must live until end of
581
   * xact, so we must make a copy of it.  Furthermore, if we're running in
582
   * serializable mode, predicate.c needs to do its own processing.
583
   */
584
0
  if (IsolationUsesXactSnapshot())
585
0
  {
586
0
    if (IsolationIsSerializable())
587
0
      SetSerializableTransactionSnapshot(CurrentSnapshot, sourcevxid,
588
0
                         sourcepid);
589
    /* Make a saved copy */
590
0
    CurrentSnapshot = CopySnapshot(CurrentSnapshot);
591
0
    FirstXactSnapshot = CurrentSnapshot;
592
    /* Mark it as "registered" in FirstXactSnapshot */
593
0
    FirstXactSnapshot->regd_count++;
594
0
    pairingheap_add(&RegisteredSnapshots, &FirstXactSnapshot->ph_node);
595
0
  }
596
597
0
  FirstSnapshotSet = true;
598
0
}
599
600
/*
601
 * CopySnapshot
602
 *    Copy the given snapshot.
603
 *
604
 * The copy is palloc'd in TopTransactionContext and has initial refcounts set
605
 * to 0.  The returned snapshot has the copied flag set.
606
 */
607
static Snapshot
608
CopySnapshot(Snapshot snapshot)
609
0
{
610
0
  Snapshot  newsnap;
611
0
  Size    subxipoff;
612
0
  Size    size;
613
614
0
  Assert(snapshot != InvalidSnapshot);
615
616
  /* We allocate any XID arrays needed in the same palloc block. */
617
0
  size = subxipoff = sizeof(SnapshotData) +
618
0
    snapshot->xcnt * sizeof(TransactionId);
619
0
  if (snapshot->subxcnt > 0)
620
0
    size += snapshot->subxcnt * sizeof(TransactionId);
621
622
0
  newsnap = (Snapshot) MemoryContextAlloc(TopTransactionContext, size);
623
0
  memcpy(newsnap, snapshot, sizeof(SnapshotData));
624
625
0
  newsnap->regd_count = 0;
626
0
  newsnap->active_count = 0;
627
0
  newsnap->copied = true;
628
0
  newsnap->snapXactCompletionCount = 0;
629
630
  /* setup XID array */
631
0
  if (snapshot->xcnt > 0)
632
0
  {
633
0
    newsnap->xip = (TransactionId *) (newsnap + 1);
634
0
    memcpy(newsnap->xip, snapshot->xip,
635
0
         snapshot->xcnt * sizeof(TransactionId));
636
0
  }
637
0
  else
638
0
    newsnap->xip = NULL;
639
640
  /*
641
   * Setup subXID array. Don't bother to copy it if it had overflowed,
642
   * though, because it's not used anywhere in that case. Except if it's a
643
   * snapshot taken during recovery; all the top-level XIDs are in subxip as
644
   * well in that case, so we mustn't lose them.
645
   */
646
0
  if (snapshot->subxcnt > 0 &&
647
0
    (!snapshot->suboverflowed || snapshot->takenDuringRecovery))
648
0
  {
649
0
    newsnap->subxip = (TransactionId *) ((char *) newsnap + subxipoff);
650
0
    memcpy(newsnap->subxip, snapshot->subxip,
651
0
         snapshot->subxcnt * sizeof(TransactionId));
652
0
  }
653
0
  else
654
0
    newsnap->subxip = NULL;
655
656
0
  return newsnap;
657
0
}
658
659
/*
660
 * FreeSnapshot
661
 *    Free the memory associated with a snapshot.
662
 */
663
static void
664
FreeSnapshot(Snapshot snapshot)
665
0
{
666
0
  Assert(snapshot->regd_count == 0);
667
0
  Assert(snapshot->active_count == 0);
668
0
  Assert(snapshot->copied);
669
670
0
  pfree(snapshot);
671
0
}
672
673
/*
674
 * PushActiveSnapshot
675
 *    Set the given snapshot as the current active snapshot
676
 *
677
 * If the passed snapshot is a statically-allocated one, or it is possibly
678
 * subject to a future command counter update, create a new long-lived copy
679
 * with active refcount=1.  Otherwise, only increment the refcount.
680
 */
681
void
682
PushActiveSnapshot(Snapshot snapshot)
683
0
{
684
0
  PushActiveSnapshotWithLevel(snapshot, GetCurrentTransactionNestLevel());
685
0
}
686
687
/*
688
 * PushActiveSnapshotWithLevel
689
 *    Set the given snapshot as the current active snapshot
690
 *
691
 * Same as PushActiveSnapshot except that caller can specify the
692
 * transaction nesting level that "owns" the snapshot.  This level
693
 * must not be deeper than the current top of the snapshot stack.
694
 */
695
void
696
PushActiveSnapshotWithLevel(Snapshot snapshot, int snap_level)
697
0
{
698
0
  ActiveSnapshotElt *newactive;
699
700
0
  Assert(snapshot != InvalidSnapshot);
701
0
  Assert(ActiveSnapshot == NULL || snap_level >= ActiveSnapshot->as_level);
702
703
0
  newactive = MemoryContextAlloc(TopTransactionContext, sizeof(ActiveSnapshotElt));
704
705
  /*
706
   * Checking SecondarySnapshot is probably useless here, but it seems
707
   * better to be sure.
708
   */
709
0
  if (snapshot == CurrentSnapshot || snapshot == SecondarySnapshot ||
710
0
    !snapshot->copied)
711
0
    newactive->as_snap = CopySnapshot(snapshot);
712
0
  else
713
0
    newactive->as_snap = snapshot;
714
715
0
  newactive->as_next = ActiveSnapshot;
716
0
  newactive->as_level = snap_level;
717
718
0
  newactive->as_snap->active_count++;
719
720
0
  ActiveSnapshot = newactive;
721
0
}
722
723
/*
724
 * PushCopiedSnapshot
725
 *    As above, except forcibly copy the presented snapshot.
726
 *
727
 * This should be used when the ActiveSnapshot has to be modifiable, for
728
 * example if the caller intends to call UpdateActiveSnapshotCommandId.
729
 * The new snapshot will be released when popped from the stack.
730
 */
731
void
732
PushCopiedSnapshot(Snapshot snapshot)
733
0
{
734
0
  PushActiveSnapshot(CopySnapshot(snapshot));
735
0
}
736
737
/*
738
 * UpdateActiveSnapshotCommandId
739
 *
740
 * Update the current CID of the active snapshot.  This can only be applied
741
 * to a snapshot that is not referenced elsewhere.
742
 */
743
void
744
UpdateActiveSnapshotCommandId(void)
745
0
{
746
0
  CommandId save_curcid,
747
0
        curcid;
748
749
0
  Assert(ActiveSnapshot != NULL);
750
0
  Assert(ActiveSnapshot->as_snap->active_count == 1);
751
0
  Assert(ActiveSnapshot->as_snap->regd_count == 0);
752
753
  /*
754
   * Don't allow modification of the active snapshot during parallel
755
   * operation.  We share the snapshot to worker backends at the beginning
756
   * of parallel operation, so any change to the snapshot can lead to
757
   * inconsistencies.  We have other defenses against
758
   * CommandCounterIncrement, but there are a few places that call this
759
   * directly, so we put an additional guard here.
760
   */
761
0
  save_curcid = ActiveSnapshot->as_snap->curcid;
762
0
  curcid = GetCurrentCommandId(false);
763
0
  if (IsInParallelMode() && save_curcid != curcid)
764
0
    elog(ERROR, "cannot modify commandid in active snapshot during a parallel operation");
765
0
  ActiveSnapshot->as_snap->curcid = curcid;
766
0
}
767
768
/*
769
 * PopActiveSnapshot
770
 *
771
 * Remove the topmost snapshot from the active snapshot stack, decrementing the
772
 * reference count, and free it if this was the last reference.
773
 */
774
void
775
PopActiveSnapshot(void)
776
0
{
777
0
  ActiveSnapshotElt *newstack;
778
779
0
  newstack = ActiveSnapshot->as_next;
780
781
0
  Assert(ActiveSnapshot->as_snap->active_count > 0);
782
783
0
  ActiveSnapshot->as_snap->active_count--;
784
785
0
  if (ActiveSnapshot->as_snap->active_count == 0 &&
786
0
    ActiveSnapshot->as_snap->regd_count == 0)
787
0
    FreeSnapshot(ActiveSnapshot->as_snap);
788
789
0
  pfree(ActiveSnapshot);
790
0
  ActiveSnapshot = newstack;
791
792
0
  SnapshotResetXmin();
793
0
}
794
795
/*
796
 * GetActiveSnapshot
797
 *    Return the topmost snapshot in the Active stack.
798
 */
799
Snapshot
800
GetActiveSnapshot(void)
801
0
{
802
0
  Assert(ActiveSnapshot != NULL);
803
804
0
  return ActiveSnapshot->as_snap;
805
0
}
806
807
/*
808
 * ActiveSnapshotSet
809
 *    Return whether there is at least one snapshot in the Active stack
810
 */
811
bool
812
ActiveSnapshotSet(void)
813
0
{
814
0
  return ActiveSnapshot != NULL;
815
0
}
816
817
/*
818
 * RegisterSnapshot
819
 *    Register a snapshot as being in use by the current resource owner
820
 *
821
 * If InvalidSnapshot is passed, it is not registered.
822
 */
823
Snapshot
824
RegisterSnapshot(Snapshot snapshot)
825
0
{
826
0
  if (snapshot == InvalidSnapshot)
827
0
    return InvalidSnapshot;
828
829
0
  return RegisterSnapshotOnOwner(snapshot, CurrentResourceOwner);
830
0
}
831
832
/*
833
 * RegisterSnapshotOnOwner
834
 *    As above, but use the specified resource owner
835
 */
836
Snapshot
837
RegisterSnapshotOnOwner(Snapshot snapshot, ResourceOwner owner)
838
0
{
839
0
  Snapshot  snap;
840
841
0
  if (snapshot == InvalidSnapshot)
842
0
    return InvalidSnapshot;
843
844
  /* Static snapshot?  Create a persistent copy */
845
0
  snap = snapshot->copied ? snapshot : CopySnapshot(snapshot);
846
847
  /* and tell resowner.c about it */
848
0
  ResourceOwnerEnlarge(owner);
849
0
  snap->regd_count++;
850
0
  ResourceOwnerRememberSnapshot(owner, snap);
851
852
0
  if (snap->regd_count == 1)
853
0
    pairingheap_add(&RegisteredSnapshots, &snap->ph_node);
854
855
0
  return snap;
856
0
}
857
858
/*
859
 * UnregisterSnapshot
860
 *
861
 * Decrement the reference count of a snapshot, remove the corresponding
862
 * reference from CurrentResourceOwner, and free the snapshot if no more
863
 * references remain.
864
 */
865
void
866
UnregisterSnapshot(Snapshot snapshot)
867
0
{
868
0
  if (snapshot == NULL)
869
0
    return;
870
871
0
  UnregisterSnapshotFromOwner(snapshot, CurrentResourceOwner);
872
0
}
873
874
/*
875
 * UnregisterSnapshotFromOwner
876
 *    As above, but use the specified resource owner
877
 */
878
void
879
UnregisterSnapshotFromOwner(Snapshot snapshot, ResourceOwner owner)
880
0
{
881
0
  if (snapshot == NULL)
882
0
    return;
883
884
0
  ResourceOwnerForgetSnapshot(owner, snapshot);
885
0
  UnregisterSnapshotNoOwner(snapshot);
886
0
}
887
888
static void
889
UnregisterSnapshotNoOwner(Snapshot snapshot)
890
0
{
891
0
  Assert(snapshot->regd_count > 0);
892
0
  Assert(!pairingheap_is_empty(&RegisteredSnapshots));
893
894
0
  snapshot->regd_count--;
895
0
  if (snapshot->regd_count == 0)
896
0
    pairingheap_remove(&RegisteredSnapshots, &snapshot->ph_node);
897
898
0
  if (snapshot->regd_count == 0 && snapshot->active_count == 0)
899
0
  {
900
0
    FreeSnapshot(snapshot);
901
0
    SnapshotResetXmin();
902
0
  }
903
0
}
904
905
/*
906
 * Comparison function for RegisteredSnapshots heap.  Snapshots are ordered
907
 * by xmin, so that the snapshot with smallest xmin is at the top.
908
 */
909
static int
910
xmin_cmp(const pairingheap_node *a, const pairingheap_node *b, void *arg)
911
0
{
912
0
  const SnapshotData *asnap = pairingheap_const_container(SnapshotData, ph_node, a);
913
0
  const SnapshotData *bsnap = pairingheap_const_container(SnapshotData, ph_node, b);
914
915
0
  if (TransactionIdPrecedes(asnap->xmin, bsnap->xmin))
916
0
    return 1;
917
0
  else if (TransactionIdFollows(asnap->xmin, bsnap->xmin))
918
0
    return -1;
919
0
  else
920
0
    return 0;
921
0
}
922
923
/*
924
 * SnapshotResetXmin
925
 *
926
 * If there are no more snapshots, we can reset our PGPROC->xmin to
927
 * InvalidTransactionId. Note we can do this without locking because we assume
928
 * that storing an Xid is atomic.
929
 *
930
 * Even if there are some remaining snapshots, we may be able to advance our
931
 * PGPROC->xmin to some degree.  This typically happens when a portal is
932
 * dropped.  For efficiency, we only consider recomputing PGPROC->xmin when
933
 * the active snapshot stack is empty; this allows us not to need to track
934
 * which active snapshot is oldest.
935
 */
936
static void
937
SnapshotResetXmin(void)
938
0
{
939
0
  Snapshot  minSnapshot;
940
941
0
  if (ActiveSnapshot != NULL)
942
0
    return;
943
944
0
  if (pairingheap_is_empty(&RegisteredSnapshots))
945
0
  {
946
0
    MyProc->xmin = TransactionXmin = InvalidTransactionId;
947
0
    return;
948
0
  }
949
950
0
  minSnapshot = pairingheap_container(SnapshotData, ph_node,
951
0
                    pairingheap_first(&RegisteredSnapshots));
952
953
0
  if (TransactionIdPrecedes(MyProc->xmin, minSnapshot->xmin))
954
0
    MyProc->xmin = TransactionXmin = minSnapshot->xmin;
955
0
}
956
957
/*
958
 * AtSubCommit_Snapshot
959
 */
960
void
961
AtSubCommit_Snapshot(int level)
962
0
{
963
0
  ActiveSnapshotElt *active;
964
965
  /*
966
   * Relabel the active snapshots set in this subtransaction as though they
967
   * are owned by the parent subxact.
968
   */
969
0
  for (active = ActiveSnapshot; active != NULL; active = active->as_next)
970
0
  {
971
0
    if (active->as_level < level)
972
0
      break;
973
0
    active->as_level = level - 1;
974
0
  }
975
0
}
976
977
/*
978
 * AtSubAbort_Snapshot
979
 *    Clean up snapshots after a subtransaction abort
980
 */
981
void
982
AtSubAbort_Snapshot(int level)
983
0
{
984
  /* Forget the active snapshots set by this subtransaction */
985
0
  while (ActiveSnapshot && ActiveSnapshot->as_level >= level)
986
0
  {
987
0
    ActiveSnapshotElt *next;
988
989
0
    next = ActiveSnapshot->as_next;
990
991
    /*
992
     * Decrement the snapshot's active count.  If it's still registered or
993
     * marked as active by an outer subtransaction, we can't free it yet.
994
     */
995
0
    Assert(ActiveSnapshot->as_snap->active_count >= 1);
996
0
    ActiveSnapshot->as_snap->active_count -= 1;
997
998
0
    if (ActiveSnapshot->as_snap->active_count == 0 &&
999
0
      ActiveSnapshot->as_snap->regd_count == 0)
1000
0
      FreeSnapshot(ActiveSnapshot->as_snap);
1001
1002
    /* and free the stack element */
1003
0
    pfree(ActiveSnapshot);
1004
1005
0
    ActiveSnapshot = next;
1006
0
  }
1007
1008
0
  SnapshotResetXmin();
1009
0
}
1010
1011
/*
1012
 * AtEOXact_Snapshot
1013
 *    Snapshot manager's cleanup function for end of transaction
1014
 */
1015
void
1016
AtEOXact_Snapshot(bool isCommit, bool resetXmin)
1017
0
{
1018
  /*
1019
   * In transaction-snapshot mode we must release our privately-managed
1020
   * reference to the transaction snapshot.  We must remove it from
1021
   * RegisteredSnapshots to keep the check below happy.  But we don't bother
1022
   * to do FreeSnapshot, for two reasons: the memory will go away with
1023
   * TopTransactionContext anyway, and if someone has left the snapshot
1024
   * stacked as active, we don't want the code below to be chasing through a
1025
   * dangling pointer.
1026
   */
1027
0
  if (FirstXactSnapshot != NULL)
1028
0
  {
1029
0
    Assert(FirstXactSnapshot->regd_count > 0);
1030
0
    Assert(!pairingheap_is_empty(&RegisteredSnapshots));
1031
0
    pairingheap_remove(&RegisteredSnapshots, &FirstXactSnapshot->ph_node);
1032
0
  }
1033
0
  FirstXactSnapshot = NULL;
1034
1035
  /*
1036
   * If we exported any snapshots, clean them up.
1037
   */
1038
0
  if (exportedSnapshots != NIL)
1039
0
  {
1040
0
    ListCell   *lc;
1041
1042
    /*
1043
     * Get rid of the files.  Unlink failure is only a WARNING because (1)
1044
     * it's too late to abort the transaction, and (2) leaving a leaked
1045
     * file around has little real consequence anyway.
1046
     *
1047
     * We also need to remove the snapshots from RegisteredSnapshots to
1048
     * prevent a warning below.
1049
     *
1050
     * As with the FirstXactSnapshot, we don't need to free resources of
1051
     * the snapshot itself as it will go away with the memory context.
1052
     */
1053
0
    foreach(lc, exportedSnapshots)
1054
0
    {
1055
0
      ExportedSnapshot *esnap = (ExportedSnapshot *) lfirst(lc);
1056
1057
0
      if (unlink(esnap->snapfile))
1058
0
        elog(WARNING, "could not unlink file \"%s\": %m",
1059
0
           esnap->snapfile);
1060
1061
0
      pairingheap_remove(&RegisteredSnapshots,
1062
0
                 &esnap->snapshot->ph_node);
1063
0
    }
1064
1065
0
    exportedSnapshots = NIL;
1066
0
  }
1067
1068
  /* Drop catalog snapshot if any */
1069
0
  InvalidateCatalogSnapshot();
1070
1071
  /* On commit, complain about leftover snapshots */
1072
0
  if (isCommit)
1073
0
  {
1074
0
    ActiveSnapshotElt *active;
1075
1076
0
    if (!pairingheap_is_empty(&RegisteredSnapshots))
1077
0
      elog(WARNING, "registered snapshots seem to remain after cleanup");
1078
1079
    /* complain about unpopped active snapshots */
1080
0
    for (active = ActiveSnapshot; active != NULL; active = active->as_next)
1081
0
      elog(WARNING, "snapshot %p still active", active);
1082
0
  }
1083
1084
  /*
1085
   * And reset our state.  We don't need to free the memory explicitly --
1086
   * it'll go away with TopTransactionContext.
1087
   */
1088
0
  ActiveSnapshot = NULL;
1089
0
  pairingheap_reset(&RegisteredSnapshots);
1090
1091
0
  CurrentSnapshot = NULL;
1092
0
  SecondarySnapshot = NULL;
1093
1094
0
  FirstSnapshotSet = false;
1095
1096
  /*
1097
   * During normal commit processing, we call ProcArrayEndTransaction() to
1098
   * reset the MyProc->xmin. That call happens prior to the call to
1099
   * AtEOXact_Snapshot(), so we need not touch xmin here at all.
1100
   */
1101
0
  if (resetXmin)
1102
0
    SnapshotResetXmin();
1103
1104
0
  Assert(resetXmin || MyProc->xmin == 0);
1105
0
}
1106
1107
1108
/*
1109
 * ExportSnapshot
1110
 *    Export the snapshot to a file so that other backends can import it.
1111
 *    Returns the token (the file name) that can be used to import this
1112
 *    snapshot.
1113
 */
1114
char *
1115
ExportSnapshot(Snapshot snapshot)
1116
0
{
1117
0
  TransactionId topXid;
1118
0
  TransactionId *children;
1119
0
  ExportedSnapshot *esnap;
1120
0
  int     nsubxids;
1121
0
  int     nchildren;
1122
0
  int     addTopXid;
1123
0
  bool    suboverflowed;
1124
0
  StringInfoData buf;
1125
0
  FILE     *f;
1126
0
  MemoryContext oldcxt;
1127
0
  char    path[MAXPGPATH];
1128
0
  char    pathtmp[MAXPGPATH];
1129
1130
  /*
1131
   * It's tempting to call RequireTransactionBlock here, since it's not very
1132
   * useful to export a snapshot that will disappear immediately afterwards.
1133
   * However, we haven't got enough information to do that, since we don't
1134
   * know if we're at top level or not.  For example, we could be inside a
1135
   * plpgsql function that is going to fire off other transactions via
1136
   * dblink.  Rather than disallow perfectly legitimate usages, don't make a
1137
   * check.
1138
   *
1139
   * Also note that we don't make any restriction on the transaction's
1140
   * isolation level; however, importers must check the level if they are
1141
   * serializable.
1142
   */
1143
1144
  /*
1145
   * Get our transaction ID if there is one, to include in the snapshot.
1146
   */
1147
0
  topXid = GetTopTransactionIdIfAny();
1148
1149
  /*
1150
   * We cannot export a snapshot from a subtransaction because there's no
1151
   * easy way for importers to verify that the same subtransaction is still
1152
   * running.
1153
   */
1154
0
  if (IsSubTransaction())
1155
0
    ereport(ERROR,
1156
0
        (errcode(ERRCODE_ACTIVE_SQL_TRANSACTION),
1157
0
         errmsg("cannot export a snapshot from a subtransaction")));
1158
1159
  /*
1160
   * We do however allow previous committed subtransactions to exist.
1161
   * Importers of the snapshot must see them as still running, so get their
1162
   * XIDs to add them to the snapshot.
1163
   */
1164
0
  nchildren = xactGetCommittedChildren(&children);
1165
1166
  /*
1167
   * We export a recovery snapshot's subxip whole (see below), so refuse an
1168
   * export that no importer will accept.  This rare edge case only happens
1169
   * when a snapshot taken during recovery is imported after its standby is
1170
   * promoted, the importing transaction subcommits many subtransactions,
1171
   * and then attempts to export the same snapshot a second time.
1172
   */
1173
0
  if (snapshot->takenDuringRecovery &&
1174
0
    snapshot->subxcnt + nchildren > GetMaxSnapshotSubxidCount())
1175
0
    ereport(ERROR,
1176
0
        (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
1177
0
         errmsg_plural("cannot export snapshot with %d running transaction ID",
1178
0
                 "cannot export snapshot with %d running transaction IDs",
1179
0
                 snapshot->subxcnt + nchildren,
1180
0
                 snapshot->subxcnt + nchildren),
1181
1182
    /*
1183
     * The singular and plural strings below are identical in English but
1184
     * could differ in translations.
1185
     */
1186
0
         errdetail_plural("A snapshot taken during recovery is exported with every transaction ID that it treats as running, and at most %d can be stored.",
1187
0
                  "A snapshot taken during recovery is exported with every transaction ID that it treats as running, and at most %d can be stored.",
1188
0
                  GetMaxSnapshotSubxidCount(),
1189
0
                  GetMaxSnapshotSubxidCount())));
1190
1191
  /*
1192
   * Generate file path for the snapshot.  We start numbering of snapshots
1193
   * inside the transaction from 1.
1194
   */
1195
0
  snprintf(path, sizeof(path), SNAPSHOT_EXPORT_DIR "/%08X-%08X-%d",
1196
0
       MyProc->vxid.procNumber, MyProc->vxid.lxid,
1197
0
       list_length(exportedSnapshots) + 1);
1198
1199
  /*
1200
   * Copy the snapshot into TopTransactionContext, add it to the
1201
   * exportedSnapshots list, and mark it pseudo-registered.  We do this to
1202
   * ensure that the snapshot's xmin is honored for the rest of the
1203
   * transaction.
1204
   */
1205
0
  snapshot = CopySnapshot(snapshot);
1206
1207
0
  oldcxt = MemoryContextSwitchTo(TopTransactionContext);
1208
0
  esnap = palloc_object(ExportedSnapshot);
1209
0
  esnap->snapfile = pstrdup(path);
1210
0
  esnap->snapshot = snapshot;
1211
0
  exportedSnapshots = lappend(exportedSnapshots, esnap);
1212
0
  MemoryContextSwitchTo(oldcxt);
1213
1214
0
  snapshot->regd_count++;
1215
0
  pairingheap_add(&RegisteredSnapshots, &snapshot->ph_node);
1216
1217
  /*
1218
   * Fill buf with a text serialization of the snapshot, plus identification
1219
   * data about this transaction.  The format expected by ImportSnapshot is
1220
   * pretty rigid: each line must be fieldname:value.
1221
   */
1222
0
  initStringInfo(&buf);
1223
1224
0
  appendStringInfo(&buf, "vxid:%d/%u\n", MyProc->vxid.procNumber, MyProc->vxid.lxid);
1225
0
  appendStringInfo(&buf, "pid:%d\n", MyProcPid);
1226
0
  appendStringInfo(&buf, "dbid:%u\n", MyDatabaseId);
1227
0
  appendStringInfo(&buf, "iso:%d\n", XactIsoLevel);
1228
0
  appendStringInfo(&buf, "ro:%d\n", XactReadOnly);
1229
1230
0
  appendStringInfo(&buf, "xmin:%u\n", snapshot->xmin);
1231
0
  appendStringInfo(&buf, "xmax:%u\n", snapshot->xmax);
1232
1233
  /*
1234
   * We must include our own top transaction ID in the top-xid data, since
1235
   * by definition we will still be running when the importing transaction
1236
   * adopts the snapshot, but GetSnapshotData never includes our own XID in
1237
   * the snapshot.  (There must, therefore, be enough room to add it.)
1238
   *
1239
   * However, it could be that our topXid is after the xmax, in which case
1240
   * we shouldn't include it because xip[] members are expected to be before
1241
   * xmax.  (We need not make the same check for subxip[] members, see
1242
   * snapshot.h.)
1243
   */
1244
0
  addTopXid = (TransactionIdIsValid(topXid) &&
1245
0
         TransactionIdPrecedes(topXid, snapshot->xmax)) ? 1 : 0;
1246
0
  appendStringInfo(&buf, "xcnt:%d\n", snapshot->xcnt + addTopXid);
1247
0
  for (uint32 i = 0; i < snapshot->xcnt; i++)
1248
0
    appendStringInfo(&buf, "xip:%u\n", snapshot->xip[i]);
1249
0
  if (addTopXid)
1250
0
    appendStringInfo(&buf, "xip:%u\n", topXid);
1251
1252
  /*
1253
   * Similarly, we add our subcommitted child XIDs to the subxid data.
1254
   *
1255
   * Report overflow when the snapshot overflowed, and also when our subxids
1256
   * won't fit in what a snapshot can hold.  For a snapshot taken outside
1257
   * recovery, claiming overflow is always safe, since it just makes
1258
   * importers fall back on pg_subtrans.
1259
   */
1260
0
  nsubxids = snapshot->subxcnt + nchildren;
1261
0
  suboverflowed = snapshot->suboverflowed ||
1262
0
    nsubxids > GetMaxSnapshotSubxidCount();
1263
1264
  /*
1265
   * Ignore the subxid array if it has overflowed, unless the snapshot was
1266
   * taken during recovery - in that case, top-level XIDs are in subxip as
1267
   * well, and we mustn't lose them.
1268
   */
1269
0
  if (suboverflowed && !snapshot->takenDuringRecovery)
1270
0
    nsubxids = 0;
1271
1272
0
  appendStringInfo(&buf, "sof:%u\n", suboverflowed);
1273
0
  appendStringInfo(&buf, "sxcnt:%d\n", nsubxids);
1274
0
  if (nsubxids > 0)
1275
0
  {
1276
0
    for (int32 i = 0; i < snapshot->subxcnt; i++)
1277
0
      appendStringInfo(&buf, "sxp:%u\n", snapshot->subxip[i]);
1278
0
    for (int32 i = 0; i < nchildren; i++)
1279
0
      appendStringInfo(&buf, "sxp:%u\n", children[i]);
1280
0
  }
1281
0
  appendStringInfo(&buf, "rec:%u\n", snapshot->takenDuringRecovery);
1282
1283
  /*
1284
   * Now write the text representation into a file.  We first write to a
1285
   * ".tmp" filename, and rename to final filename if no error.  This
1286
   * ensures that no other backend can read an incomplete file
1287
   * (ImportSnapshot won't allow it because of its valid-characters check).
1288
   */
1289
0
  snprintf(pathtmp, sizeof(pathtmp), "%s.tmp", path);
1290
0
  if (!(f = AllocateFile(pathtmp, PG_BINARY_W)))
1291
0
    ereport(ERROR,
1292
0
        (errcode_for_file_access(),
1293
0
         errmsg("could not create file \"%s\": %m", pathtmp)));
1294
1295
0
  if (fwrite(buf.data, buf.len, 1, f) != 1)
1296
0
    ereport(ERROR,
1297
0
        (errcode_for_file_access(),
1298
0
         errmsg("could not write to file \"%s\": %m", pathtmp)));
1299
1300
  /* no fsync() since file need not survive a system crash */
1301
1302
0
  if (FreeFile(f))
1303
0
    ereport(ERROR,
1304
0
        (errcode_for_file_access(),
1305
0
         errmsg("could not write to file \"%s\": %m", pathtmp)));
1306
1307
  /*
1308
   * Now that we have written everything into a .tmp file, rename the file
1309
   * to remove the .tmp suffix.
1310
   */
1311
0
  if (rename(pathtmp, path) < 0)
1312
0
    ereport(ERROR,
1313
0
        (errcode_for_file_access(),
1314
0
         errmsg("could not rename file \"%s\" to \"%s\": %m",
1315
0
            pathtmp, path)));
1316
1317
  /*
1318
   * The basename of the file is what we return from pg_export_snapshot().
1319
   * It's already in path in a textual format and we know that the path
1320
   * starts with SNAPSHOT_EXPORT_DIR.  Skip over the prefix and the slash
1321
   * and pstrdup it so as not to return the address of a local variable.
1322
   */
1323
0
  return pstrdup(path + strlen(SNAPSHOT_EXPORT_DIR) + 1);
1324
0
}
1325
1326
/*
1327
 * pg_export_snapshot
1328
 *    SQL-callable wrapper for ExportSnapshot.
1329
 */
1330
Datum
1331
pg_export_snapshot(PG_FUNCTION_ARGS)
1332
0
{
1333
0
  char     *snapshotName;
1334
1335
0
  snapshotName = ExportSnapshot(GetActiveSnapshot());
1336
0
  PG_RETURN_TEXT_P(cstring_to_text(snapshotName));
1337
0
}
1338
1339
1340
/*
1341
 * Parsing subroutines for ImportSnapshot: parse a line with the given
1342
 * prefix followed by a value, and advance *s to the next line.  The
1343
 * filename is provided for use in error messages.
1344
 */
1345
static int
1346
parseIntFromText(const char *prefix, char **s, const char *filename)
1347
0
{
1348
0
  char     *ptr = *s;
1349
0
  int     prefixlen = strlen(prefix);
1350
0
  int     val;
1351
1352
0
  if (strncmp(ptr, prefix, prefixlen) != 0)
1353
0
    ereport(ERROR,
1354
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1355
0
         errmsg("invalid snapshot data in file \"%s\"", filename)));
1356
0
  ptr += prefixlen;
1357
0
  if (sscanf(ptr, "%d", &val) != 1)
1358
0
    ereport(ERROR,
1359
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1360
0
         errmsg("invalid snapshot data in file \"%s\"", filename)));
1361
0
  ptr = strchr(ptr, '\n');
1362
0
  if (!ptr)
1363
0
    ereport(ERROR,
1364
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1365
0
         errmsg("invalid snapshot data in file \"%s\"", filename)));
1366
0
  *s = ptr + 1;
1367
0
  return val;
1368
0
}
1369
1370
static TransactionId
1371
parseXidFromText(const char *prefix, char **s, const char *filename)
1372
0
{
1373
0
  char     *ptr = *s;
1374
0
  int     prefixlen = strlen(prefix);
1375
0
  TransactionId val;
1376
1377
0
  if (strncmp(ptr, prefix, prefixlen) != 0)
1378
0
    ereport(ERROR,
1379
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1380
0
         errmsg("invalid snapshot data in file \"%s\"", filename)));
1381
0
  ptr += prefixlen;
1382
0
  if (sscanf(ptr, "%u", &val) != 1)
1383
0
    ereport(ERROR,
1384
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1385
0
         errmsg("invalid snapshot data in file \"%s\"", filename)));
1386
0
  ptr = strchr(ptr, '\n');
1387
0
  if (!ptr)
1388
0
    ereport(ERROR,
1389
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1390
0
         errmsg("invalid snapshot data in file \"%s\"", filename)));
1391
0
  *s = ptr + 1;
1392
0
  return val;
1393
0
}
1394
1395
static void
1396
parseVxidFromText(const char *prefix, char **s, const char *filename,
1397
          VirtualTransactionId *vxid)
1398
0
{
1399
0
  char     *ptr = *s;
1400
0
  int     prefixlen = strlen(prefix);
1401
1402
0
  if (strncmp(ptr, prefix, prefixlen) != 0)
1403
0
    ereport(ERROR,
1404
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1405
0
         errmsg("invalid snapshot data in file \"%s\"", filename)));
1406
0
  ptr += prefixlen;
1407
0
  if (sscanf(ptr, "%d/%u", &vxid->procNumber, &vxid->localTransactionId) != 2)
1408
0
    ereport(ERROR,
1409
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1410
0
         errmsg("invalid snapshot data in file \"%s\"", filename)));
1411
0
  ptr = strchr(ptr, '\n');
1412
0
  if (!ptr)
1413
0
    ereport(ERROR,
1414
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1415
0
         errmsg("invalid snapshot data in file \"%s\"", filename)));
1416
0
  *s = ptr + 1;
1417
0
}
1418
1419
/*
1420
 * ImportSnapshot
1421
 *    Import a previously exported snapshot.  The argument should be a
1422
 *    filename in SNAPSHOT_EXPORT_DIR.  Load the snapshot from that file.
1423
 *    This is called by "SET TRANSACTION SNAPSHOT 'foo'".
1424
 */
1425
void
1426
ImportSnapshot(const char *idstr)
1427
0
{
1428
0
  char    path[MAXPGPATH];
1429
0
  FILE     *f;
1430
0
  struct stat stat_buf;
1431
0
  char     *filebuf;
1432
0
  int     xcnt;
1433
0
  int     i;
1434
0
  VirtualTransactionId src_vxid;
1435
0
  int     src_pid;
1436
0
  Oid     src_dbid;
1437
0
  int     src_isolevel;
1438
0
  bool    src_readonly;
1439
0
  SnapshotData snapshot;
1440
1441
  /*
1442
   * Must be at top level of a fresh transaction.  Note in particular that
1443
   * we check we haven't acquired an XID --- if we have, it's conceivable
1444
   * that the snapshot would show it as not running, making for very screwy
1445
   * behavior.
1446
   */
1447
0
  if (FirstSnapshotSet ||
1448
0
    GetTopTransactionIdIfAny() != InvalidTransactionId ||
1449
0
    IsSubTransaction())
1450
0
    ereport(ERROR,
1451
0
        (errcode(ERRCODE_ACTIVE_SQL_TRANSACTION),
1452
0
         errmsg("SET TRANSACTION SNAPSHOT must be called before any query")));
1453
1454
  /*
1455
   * If we are in read committed mode then the next query would execute with
1456
   * a new snapshot thus making this function call quite useless.
1457
   */
1458
0
  if (!IsolationUsesXactSnapshot())
1459
0
    ereport(ERROR,
1460
0
        (errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
1461
0
         errmsg("a snapshot-importing transaction must have isolation level SERIALIZABLE or REPEATABLE READ")));
1462
1463
  /*
1464
   * Verify the identifier: only 0-9, A-F and hyphens are allowed.  We do
1465
   * this mainly to prevent reading arbitrary files.
1466
   */
1467
0
  if (strspn(idstr, "0123456789ABCDEF-") != strlen(idstr))
1468
0
    ereport(ERROR,
1469
0
        (errcode(ERRCODE_INVALID_PARAMETER_VALUE),
1470
0
         errmsg("invalid snapshot identifier: \"%s\"", idstr)));
1471
1472
  /* OK, read the file */
1473
0
  snprintf(path, MAXPGPATH, SNAPSHOT_EXPORT_DIR "/%s", idstr);
1474
1475
0
  f = AllocateFile(path, PG_BINARY_R);
1476
0
  if (!f)
1477
0
  {
1478
    /*
1479
     * If file is missing while identifier has a correct format, avoid
1480
     * system errors.
1481
     */
1482
0
    if (errno == ENOENT)
1483
0
      ereport(ERROR,
1484
0
          (errcode(ERRCODE_UNDEFINED_OBJECT),
1485
0
           errmsg("snapshot \"%s\" does not exist", idstr)));
1486
0
    else
1487
0
      ereport(ERROR,
1488
0
          (errcode_for_file_access(),
1489
0
           errmsg("could not open file \"%s\" for reading: %m",
1490
0
              path)));
1491
0
  }
1492
1493
  /* get the size of the file so that we know how much memory we need */
1494
0
  if (fstat(fileno(f), &stat_buf))
1495
0
    elog(ERROR, "could not stat file \"%s\": %m", path);
1496
1497
  /* and read the file into a palloc'd string */
1498
0
  filebuf = (char *) palloc(stat_buf.st_size + 1);
1499
0
  if (fread(filebuf, stat_buf.st_size, 1, f) != 1)
1500
0
    elog(ERROR, "could not read file \"%s\": %m", path);
1501
1502
0
  filebuf[stat_buf.st_size] = '\0';
1503
1504
0
  FreeFile(f);
1505
1506
  /*
1507
   * Construct a snapshot struct by parsing the file content.
1508
   */
1509
0
  memset(&snapshot, 0, sizeof(snapshot));
1510
1511
0
  parseVxidFromText("vxid:", &filebuf, path, &src_vxid);
1512
0
  src_pid = parseIntFromText("pid:", &filebuf, path);
1513
  /* we abuse parseXidFromText a bit here ... */
1514
0
  src_dbid = parseXidFromText("dbid:", &filebuf, path);
1515
0
  src_isolevel = parseIntFromText("iso:", &filebuf, path);
1516
0
  src_readonly = parseIntFromText("ro:", &filebuf, path);
1517
1518
0
  snapshot.snapshot_type = SNAPSHOT_MVCC;
1519
1520
0
  snapshot.xmin = parseXidFromText("xmin:", &filebuf, path);
1521
0
  snapshot.xmax = parseXidFromText("xmax:", &filebuf, path);
1522
1523
0
  snapshot.xcnt = xcnt = parseIntFromText("xcnt:", &filebuf, path);
1524
1525
  /* sanity-check the xid count before palloc */
1526
0
  if (xcnt < 0 || xcnt > GetMaxSnapshotXidCount())
1527
0
    ereport(ERROR,
1528
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1529
0
         errmsg("invalid snapshot data in file \"%s\"", path)));
1530
1531
0
  snapshot.xip = palloc_array(TransactionId, xcnt);
1532
0
  for (i = 0; i < xcnt; i++)
1533
0
    snapshot.xip[i] = parseXidFromText("xip:", &filebuf, path);
1534
1535
0
  snapshot.suboverflowed = parseIntFromText("sof:", &filebuf, path);
1536
0
  snapshot.subxcnt = xcnt = parseIntFromText("sxcnt:", &filebuf, path);
1537
0
  snapshot.subxip = NULL;
1538
1539
0
  if (snapshot.subxcnt)
1540
0
  {
1541
    /* sanity-check the xid count before palloc */
1542
0
    if (xcnt < 0 || xcnt > GetMaxSnapshotSubxidCount())
1543
0
      ereport(ERROR,
1544
0
          (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1545
0
           errmsg("invalid snapshot data in file \"%s\"", path)));
1546
1547
0
    snapshot.subxip = palloc_array(TransactionId, xcnt);
1548
0
    for (i = 0; i < xcnt; i++)
1549
0
      snapshot.subxip[i] = parseXidFromText("sxp:", &filebuf, path);
1550
0
  }
1551
1552
0
  snapshot.takenDuringRecovery = parseIntFromText("rec:", &filebuf, path);
1553
1554
  /*
1555
   * Do some additional sanity checking, just to protect ourselves.  We
1556
   * don't trouble to check the array elements, just the most critical
1557
   * fields.
1558
   */
1559
0
  if (!VirtualTransactionIdIsValid(src_vxid) ||
1560
0
    !OidIsValid(src_dbid) ||
1561
0
    !TransactionIdIsNormal(snapshot.xmin) ||
1562
0
    !TransactionIdIsNormal(snapshot.xmax))
1563
0
    ereport(ERROR,
1564
0
        (errcode(ERRCODE_INVALID_TEXT_REPRESENTATION),
1565
0
         errmsg("invalid snapshot data in file \"%s\"", path)));
1566
1567
  /*
1568
   * If we're serializable, the source transaction must be too, otherwise
1569
   * predicate.c has problems (SxactGlobalXmin could go backwards).  Also, a
1570
   * non-read-only transaction can't adopt a snapshot from a read-only
1571
   * transaction, as predicate.c handles the cases very differently.
1572
   */
1573
0
  if (IsolationIsSerializable())
1574
0
  {
1575
0
    if (src_isolevel != XACT_SERIALIZABLE)
1576
0
      ereport(ERROR,
1577
0
          (errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
1578
0
           errmsg("a serializable transaction cannot import a snapshot from a non-serializable transaction")));
1579
0
    if (src_readonly && !XactReadOnly)
1580
0
      ereport(ERROR,
1581
0
          (errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
1582
0
           errmsg("a non-read-only serializable transaction cannot import a snapshot from a read-only transaction")));
1583
0
  }
1584
1585
  /*
1586
   * We cannot import a snapshot that was taken in a different database,
1587
   * because vacuum calculates OldestXmin on a per-database basis; so the
1588
   * source transaction's xmin doesn't protect us from data loss.  This
1589
   * restriction could be removed if the source transaction were to mark its
1590
   * xmin as being globally applicable.  But that would require some
1591
   * additional syntax, since that has to be known when the snapshot is
1592
   * initially taken.  (See pgsql-hackers discussion of 2011-10-21.)
1593
   */
1594
0
  if (src_dbid != MyDatabaseId)
1595
0
    ereport(ERROR,
1596
0
        (errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
1597
0
         errmsg("cannot import a snapshot from a different database")));
1598
1599
  /* OK, install the snapshot */
1600
0
  SetTransactionSnapshot(&snapshot, &src_vxid, src_pid, NULL);
1601
0
}
1602
1603
/*
1604
 * XactHasExportedSnapshots
1605
 *    Test whether current transaction has exported any snapshots.
1606
 */
1607
bool
1608
XactHasExportedSnapshots(void)
1609
0
{
1610
0
  return (exportedSnapshots != NIL);
1611
0
}
1612
1613
/*
1614
 * DeleteAllExportedSnapshotFiles
1615
 *    Clean up any files that have been left behind by a crashed backend
1616
 *    that had exported snapshots before it died.
1617
 *
1618
 * This should be called during database startup or crash recovery.
1619
 */
1620
void
1621
DeleteAllExportedSnapshotFiles(void)
1622
{
1623
  char    buf[MAXPGPATH + sizeof(SNAPSHOT_EXPORT_DIR)];
1624
  DIR      *s_dir;
1625
  struct dirent *s_de;
1626
1627
  /*
1628
   * Problems in reading the directory, or unlinking files, are reported at
1629
   * LOG level.  Since we're running in the startup process, ERROR level
1630
   * would prevent database start, and it's not important enough for that.
1631
   */
1632
  s_dir = AllocateDir(SNAPSHOT_EXPORT_DIR);
1633
1634
  while ((s_de = ReadDirExtended(s_dir, SNAPSHOT_EXPORT_DIR, LOG)) != NULL)
1635
  {
1636
    if (strcmp(s_de->d_name, ".") == 0 ||
1637
      strcmp(s_de->d_name, "..") == 0)
1638
      continue;
1639
1640
    snprintf(buf, sizeof(buf), SNAPSHOT_EXPORT_DIR "/%s", s_de->d_name);
1641
1642
    if (unlink(buf) != 0)
1643
      ereport(LOG,
1644
          (errcode_for_file_access(),
1645
           errmsg("could not remove file \"%s\": %m", buf)));
1646
  }
1647
1648
  FreeDir(s_dir);
1649
}
1650
1651
/*
1652
 * ThereAreNoPriorRegisteredSnapshots
1653
 *    Is the registered snapshot count less than or equal to one?
1654
 *
1655
 * Don't use this to settle important decisions.  While zero registrations and
1656
 * no ActiveSnapshot would confirm a certain idleness, the system makes no
1657
 * guarantees about the significance of one registered snapshot.
1658
 */
1659
bool
1660
ThereAreNoPriorRegisteredSnapshots(void)
1661
0
{
1662
0
  if (pairingheap_is_empty(&RegisteredSnapshots) ||
1663
0
    pairingheap_is_singular(&RegisteredSnapshots))
1664
0
    return true;
1665
1666
0
  return false;
1667
0
}
1668
1669
/*
1670
 * HaveRegisteredOrActiveSnapshot
1671
 *    Is there any registered or active snapshot?
1672
 *
1673
 * NB: Unless pushed or active, the cached catalog snapshot will not cause
1674
 * this function to return true. That allows this function to be used in
1675
 * checks enforcing a longer-lived snapshot.
1676
 */
1677
bool
1678
HaveRegisteredOrActiveSnapshot(void)
1679
0
{
1680
0
  if (ActiveSnapshot != NULL)
1681
0
    return true;
1682
1683
  /*
1684
   * The catalog snapshot is in RegisteredSnapshots when valid, but can be
1685
   * removed at any time due to invalidation processing. If explicitly
1686
   * registered more than one snapshot has to be in RegisteredSnapshots.
1687
   */
1688
0
  if (CatalogSnapshot != NULL &&
1689
0
    pairingheap_is_singular(&RegisteredSnapshots))
1690
0
    return false;
1691
1692
0
  return !pairingheap_is_empty(&RegisteredSnapshots);
1693
0
}
1694
1695
1696
/*
1697
 * Setup a snapshot that replaces normal catalog snapshots that allows catalog
1698
 * access to behave just like it did at a certain point in the past.
1699
 *
1700
 * Needed for logical decoding.
1701
 */
1702
void
1703
SetupHistoricSnapshot(Snapshot historic_snapshot, HTAB *tuplecids)
1704
0
{
1705
0
  Assert(historic_snapshot != NULL);
1706
1707
  /* setup the timetravel snapshot */
1708
0
  HistoricSnapshot = historic_snapshot;
1709
1710
  /* setup (cmin, cmax) lookup hash */
1711
0
  tuplecid_data = tuplecids;
1712
0
}
1713
1714
1715
/*
1716
 * Make catalog snapshots behave normally again.
1717
 */
1718
void
1719
TeardownHistoricSnapshot(bool is_error)
1720
0
{
1721
0
  HistoricSnapshot = NULL;
1722
0
  tuplecid_data = NULL;
1723
0
}
1724
1725
bool
1726
HistoricSnapshotActive(void)
1727
0
{
1728
0
  return HistoricSnapshot != NULL;
1729
0
}
1730
1731
HTAB *
1732
HistoricSnapshotGetTupleCids(void)
1733
0
{
1734
0
  Assert(HistoricSnapshotActive());
1735
0
  return tuplecid_data;
1736
0
}
1737
1738
/*
1739
 * EstimateSnapshotSpace
1740
 *    Returns the size needed to store the given snapshot.
1741
 *
1742
 * We are exporting only required fields from the Snapshot, stored in
1743
 * SerializedSnapshotData.
1744
 */
1745
Size
1746
EstimateSnapshotSpace(Snapshot snapshot)
1747
0
{
1748
0
  Size    size;
1749
1750
0
  Assert(snapshot != InvalidSnapshot);
1751
0
  Assert(snapshot->snapshot_type == SNAPSHOT_MVCC);
1752
1753
  /* We allocate any XID arrays needed in the same palloc block. */
1754
0
  size = add_size(sizeof(SerializedSnapshotData),
1755
0
          mul_size(snapshot->xcnt, sizeof(TransactionId)));
1756
0
  if (snapshot->subxcnt > 0 &&
1757
0
    (!snapshot->suboverflowed || snapshot->takenDuringRecovery))
1758
0
    size = add_size(size,
1759
0
            mul_size(snapshot->subxcnt, sizeof(TransactionId)));
1760
1761
0
  return size;
1762
0
}
1763
1764
/*
1765
 * SerializeSnapshot
1766
 *    Dumps the serialized snapshot (extracted from given snapshot) onto the
1767
 *    memory location at start_address.
1768
 */
1769
void
1770
SerializeSnapshot(Snapshot snapshot, char *start_address)
1771
0
{
1772
0
  SerializedSnapshotData serialized_snapshot = {0};
1773
1774
0
  Assert(snapshot->subxcnt >= 0);
1775
1776
  /* Copy all required fields */
1777
0
  serialized_snapshot.xmin = snapshot->xmin;
1778
0
  serialized_snapshot.xmax = snapshot->xmax;
1779
0
  serialized_snapshot.xcnt = snapshot->xcnt;
1780
0
  serialized_snapshot.subxcnt = snapshot->subxcnt;
1781
0
  serialized_snapshot.suboverflowed = snapshot->suboverflowed;
1782
0
  serialized_snapshot.takenDuringRecovery = snapshot->takenDuringRecovery;
1783
0
  serialized_snapshot.curcid = snapshot->curcid;
1784
1785
  /*
1786
   * Ignore the SubXID array if it has overflowed, unless the snapshot was
1787
   * taken during recovery - in that case, top-level XIDs are in subxip as
1788
   * well, and we mustn't lose them.
1789
   */
1790
0
  if (serialized_snapshot.suboverflowed && !snapshot->takenDuringRecovery)
1791
0
    serialized_snapshot.subxcnt = 0;
1792
1793
  /* Copy struct to possibly-unaligned buffer */
1794
0
  memcpy(start_address,
1795
0
       &serialized_snapshot, sizeof(SerializedSnapshotData));
1796
1797
  /* Copy XID array */
1798
0
  if (snapshot->xcnt > 0)
1799
0
    memcpy((TransactionId *) (start_address +
1800
0
                  sizeof(SerializedSnapshotData)),
1801
0
         snapshot->xip, snapshot->xcnt * sizeof(TransactionId));
1802
1803
  /*
1804
   * Copy SubXID array. Don't bother to copy it if it had overflowed,
1805
   * though, because it's not used anywhere in that case. Except if it's a
1806
   * snapshot taken during recovery; all the top-level XIDs are in subxip as
1807
   * well in that case, so we mustn't lose them.
1808
   */
1809
0
  if (serialized_snapshot.subxcnt > 0)
1810
0
  {
1811
0
    Size    subxipoff = sizeof(SerializedSnapshotData) +
1812
0
      snapshot->xcnt * sizeof(TransactionId);
1813
1814
0
    memcpy((TransactionId *) (start_address + subxipoff),
1815
0
         snapshot->subxip, snapshot->subxcnt * sizeof(TransactionId));
1816
0
  }
1817
0
}
1818
1819
/*
1820
 * RestoreSnapshot
1821
 *    Restore a serialized snapshot from the specified address.
1822
 *
1823
 * The copy is palloc'd in TopTransactionContext and has initial refcounts set
1824
 * to 0.  The returned snapshot has the copied flag set.
1825
 */
1826
Snapshot
1827
RestoreSnapshot(char *start_address)
1828
0
{
1829
0
  SerializedSnapshotData serialized_snapshot;
1830
0
  Size    size;
1831
0
  Snapshot  snapshot;
1832
0
  TransactionId *serialized_xids;
1833
1834
0
  memcpy(&serialized_snapshot, start_address,
1835
0
       sizeof(SerializedSnapshotData));
1836
0
  serialized_xids = (TransactionId *)
1837
0
    (start_address + sizeof(SerializedSnapshotData));
1838
1839
  /* We allocate any XID arrays needed in the same palloc block. */
1840
0
  size = sizeof(SnapshotData)
1841
0
    + serialized_snapshot.xcnt * sizeof(TransactionId)
1842
0
    + serialized_snapshot.subxcnt * sizeof(TransactionId);
1843
1844
  /* Copy all required fields */
1845
0
  snapshot = (Snapshot) MemoryContextAlloc(TopTransactionContext, size);
1846
0
  snapshot->snapshot_type = SNAPSHOT_MVCC;
1847
0
  snapshot->xmin = serialized_snapshot.xmin;
1848
0
  snapshot->xmax = serialized_snapshot.xmax;
1849
0
  snapshot->xip = NULL;
1850
0
  snapshot->xcnt = serialized_snapshot.xcnt;
1851
0
  snapshot->subxip = NULL;
1852
0
  snapshot->subxcnt = serialized_snapshot.subxcnt;
1853
0
  snapshot->suboverflowed = serialized_snapshot.suboverflowed;
1854
0
  snapshot->takenDuringRecovery = serialized_snapshot.takenDuringRecovery;
1855
0
  snapshot->curcid = serialized_snapshot.curcid;
1856
0
  snapshot->snapXactCompletionCount = 0;
1857
1858
  /* Copy XIDs, if present. */
1859
0
  if (serialized_snapshot.xcnt > 0)
1860
0
  {
1861
0
    snapshot->xip = (TransactionId *) (snapshot + 1);
1862
0
    memcpy(snapshot->xip, serialized_xids,
1863
0
         serialized_snapshot.xcnt * sizeof(TransactionId));
1864
0
  }
1865
1866
  /* Copy SubXIDs, if present. */
1867
0
  if (serialized_snapshot.subxcnt > 0)
1868
0
  {
1869
0
    snapshot->subxip = ((TransactionId *) (snapshot + 1)) +
1870
0
      serialized_snapshot.xcnt;
1871
0
    memcpy(snapshot->subxip, serialized_xids + serialized_snapshot.xcnt,
1872
0
         serialized_snapshot.subxcnt * sizeof(TransactionId));
1873
0
  }
1874
1875
  /* Set the copied flag so that the caller will set refcounts correctly. */
1876
0
  snapshot->regd_count = 0;
1877
0
  snapshot->active_count = 0;
1878
0
  snapshot->copied = true;
1879
1880
0
  return snapshot;
1881
0
}
1882
1883
/*
1884
 * Install a restored snapshot as the transaction snapshot.
1885
 */
1886
void
1887
RestoreTransactionSnapshot(Snapshot snapshot, PGPROC *source_pgproc)
1888
0
{
1889
0
  SetTransactionSnapshot(snapshot, NULL, InvalidPid, source_pgproc);
1890
0
}
1891
1892
/*
1893
 * XidInMVCCSnapshot
1894
 *    Is the given XID still-in-progress according to the snapshot?
1895
 *
1896
 * Note: GetSnapshotData never stores either top xid or subxids of our own
1897
 * backend into a snapshot, so these xids will not be reported as "running"
1898
 * by this function.  This is OK for current uses, because we always check
1899
 * TransactionIdIsCurrentTransactionId first, except when it's known the
1900
 * XID could not be ours anyway.
1901
 */
1902
bool
1903
XidInMVCCSnapshot(TransactionId xid, Snapshot snapshot)
1904
0
{
1905
  /*
1906
   * Make a quick range check to eliminate most XIDs without looking at the
1907
   * xip arrays.  Note that this is OK even if we convert a subxact XID to
1908
   * its parent below, because a subxact with XID < xmin has surely also got
1909
   * a parent with XID < xmin, while one with XID >= xmax must belong to a
1910
   * parent that was not yet committed at the time of this snapshot.
1911
   */
1912
1913
  /* Any xid < xmin is not in-progress */
1914
0
  if (TransactionIdPrecedes(xid, snapshot->xmin))
1915
0
    return false;
1916
  /* Any xid >= xmax is in-progress */
1917
0
  if (TransactionIdFollowsOrEquals(xid, snapshot->xmax))
1918
0
    return true;
1919
1920
  /*
1921
   * Snapshot information is stored slightly differently in snapshots taken
1922
   * during recovery.
1923
   */
1924
0
  if (!snapshot->takenDuringRecovery)
1925
0
  {
1926
    /*
1927
     * If the snapshot contains full subxact data, the fastest way to
1928
     * check things is just to compare the given XID against both subxact
1929
     * XIDs and top-level XIDs.  If the snapshot overflowed, we have to
1930
     * use pg_subtrans to convert a subxact XID to its parent XID, but
1931
     * then we need only look at top-level XIDs not subxacts.
1932
     */
1933
0
    if (!snapshot->suboverflowed)
1934
0
    {
1935
      /* we have full data, so search subxip */
1936
0
      if (pg_lfind32(xid, snapshot->subxip, snapshot->subxcnt))
1937
0
        return true;
1938
1939
      /* not there, fall through to search xip[] */
1940
0
    }
1941
0
    else
1942
0
    {
1943
      /*
1944
       * Snapshot overflowed, so convert xid to top-level.  This is safe
1945
       * because we eliminated too-old XIDs above.
1946
       */
1947
0
      xid = SubTransGetTopmostTransaction(xid);
1948
1949
      /*
1950
       * If xid was indeed a subxact, we might now have an xid < xmin,
1951
       * so recheck to avoid an array scan.  No point in rechecking
1952
       * xmax.
1953
       */
1954
0
      if (TransactionIdPrecedes(xid, snapshot->xmin))
1955
0
        return false;
1956
0
    }
1957
1958
0
    if (pg_lfind32(xid, snapshot->xip, snapshot->xcnt))
1959
0
      return true;
1960
0
  }
1961
0
  else
1962
0
  {
1963
    /*
1964
     * In recovery we store all xids in the subxip array because it is by
1965
     * far the bigger array, and we mostly don't know which xids are
1966
     * top-level and which are subxacts. The xip array is empty.
1967
     *
1968
     * We start by searching subtrans, if we overflowed.
1969
     */
1970
0
    if (snapshot->suboverflowed)
1971
0
    {
1972
      /*
1973
       * Snapshot overflowed, so convert xid to top-level.  This is safe
1974
       * because we eliminated too-old XIDs above.
1975
       */
1976
0
      xid = SubTransGetTopmostTransaction(xid);
1977
1978
      /*
1979
       * If xid was indeed a subxact, we might now have an xid < xmin,
1980
       * so recheck to avoid an array scan.  No point in rechecking
1981
       * xmax.
1982
       */
1983
0
      if (TransactionIdPrecedes(xid, snapshot->xmin))
1984
0
        return false;
1985
0
    }
1986
1987
    /*
1988
     * We now have either a top-level xid higher than xmin or an
1989
     * indeterminate xid. We don't know whether it's top level or subxact
1990
     * but it doesn't matter. If it's present, the xid is visible.
1991
     */
1992
0
    if (pg_lfind32(xid, snapshot->subxip, snapshot->subxcnt))
1993
0
      return true;
1994
0
  }
1995
1996
0
  return false;
1997
0
}
1998
1999
/* ResourceOwner callbacks */
2000
2001
static void
2002
ResOwnerReleaseSnapshot(Datum res)
2003
0
{
2004
0
  UnregisterSnapshotNoOwner((Snapshot) DatumGetPointer(res));
2005
0
}