Coverage Report

Created: 2026-08-13 07:12

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/postgres/src/backend/access/gin/ginfast.c
Line
Count
Source
1
/*-------------------------------------------------------------------------
2
 *
3
 * ginfast.c
4
 *    Fast insert routines for the Postgres inverted index access method.
5
 *    Pending entries are stored in linear list of pages.  Later on
6
 *    (typically during VACUUM), ginInsertCleanup() will be invoked to
7
 *    transfer pending entries into the regular index structure.  This
8
 *    wins because bulk insertion is much more efficient than retail.
9
 *
10
 * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
11
 * Portions Copyright (c) 1994, Regents of the University of California
12
 *
13
 * IDENTIFICATION
14
 *      src/backend/access/gin/ginfast.c
15
 *
16
 *-------------------------------------------------------------------------
17
 */
18
19
#include "postgres.h"
20
21
#include "access/gin_private.h"
22
#include "access/ginxlog.h"
23
#include "access/xlog.h"
24
#include "access/xloginsert.h"
25
#include "catalog/pg_am.h"
26
#include "commands/vacuum.h"
27
#include "miscadmin.h"
28
#include "port/pg_bitutils.h"
29
#include "postmaster/autovacuum.h"
30
#include "storage/indexfsm.h"
31
#include "storage/lmgr.h"
32
#include "storage/predicate.h"
33
#include "utils/acl.h"
34
#include "utils/fmgrprotos.h"
35
#include "utils/memutils.h"
36
#include "utils/rel.h"
37
38
/* GUC parameter */
39
int     gin_pending_list_limit = 0;
40
41
#define GIN_PAGE_FREESIZE \
42
0
  ( (Size) BLCKSZ - MAXALIGN(SizeOfPageHeaderData) - MAXALIGN(sizeof(GinPageOpaqueData)) )
43
44
typedef struct KeyArray
45
{
46
  Datum    *keys;     /* expansible array */
47
  GinNullCategory *categories;  /* another expansible array */
48
  int32   nvalues;    /* current number of valid entries */
49
  int32   maxvalues;    /* allocated size of arrays */
50
} KeyArray;
51
52
53
/*
54
 * Build a pending-list page from the given array of tuples, and write it out.
55
 *
56
 * Returns amount of free space left on the page.
57
 */
58
static int32
59
writeListPage(Relation index, Buffer buffer,
60
        const IndexTuple *tuples, int32 ntuples, BlockNumber rightlink)
61
0
{
62
0
  Page    page = BufferGetPage(buffer);
63
0
  int32   i,
64
0
        freesize,
65
0
        size = 0;
66
0
  OffsetNumber l,
67
0
        off;
68
0
  PGAlignedBlock workspace;
69
0
  char     *ptr;
70
71
0
  START_CRIT_SECTION();
72
73
0
  GinInitBuffer(buffer, GIN_LIST);
74
75
0
  off = FirstOffsetNumber;
76
0
  ptr = workspace.data;
77
78
0
  for (i = 0; i < ntuples; i++)
79
0
  {
80
0
    int     this_size = IndexTupleSize(tuples[i]);
81
82
0
    memcpy(ptr, tuples[i], this_size);
83
0
    ptr += this_size;
84
0
    size += this_size;
85
86
0
    l = PageAddItem(page, tuples[i], this_size, off, false, false);
87
88
0
    if (l == InvalidOffsetNumber)
89
0
      elog(ERROR, "failed to add item to index page in \"%s\"",
90
0
         RelationGetRelationName(index));
91
92
0
    off++;
93
0
  }
94
95
0
  Assert(size <= BLCKSZ);   /* else we overran workspace */
96
97
0
  GinPageGetOpaque(page)->rightlink = rightlink;
98
99
  /*
100
   * tail page may contain only whole row(s) or final part of row placed on
101
   * previous pages (a "row" here meaning all the index tuples generated for
102
   * one heap tuple)
103
   */
104
0
  if (rightlink == InvalidBlockNumber)
105
0
  {
106
0
    GinPageSetFullRow(page);
107
0
    GinPageGetOpaque(page)->maxoff = 1;
108
0
  }
109
0
  else
110
0
  {
111
0
    GinPageGetOpaque(page)->maxoff = 0;
112
0
  }
113
114
0
  MarkBufferDirty(buffer);
115
116
0
  if (RelationNeedsWAL(index))
117
0
  {
118
0
    ginxlogInsertListPage data;
119
0
    XLogRecPtr  recptr;
120
121
0
    data.rightlink = rightlink;
122
0
    data.ntuples = ntuples;
123
124
0
    XLogBeginInsert();
125
0
    XLogRegisterData(&data, sizeof(ginxlogInsertListPage));
126
127
0
    XLogRegisterBuffer(0, buffer, REGBUF_WILL_INIT);
128
0
    XLogRegisterBufData(0, workspace.data, size);
129
130
0
    recptr = XLogInsert(RM_GIN_ID, XLOG_GIN_INSERT_LISTPAGE);
131
0
    PageSetLSN(page, recptr);
132
0
  }
133
134
  /* get free space before releasing buffer */
135
0
  freesize = PageGetExactFreeSpace(page);
136
137
0
  END_CRIT_SECTION();
138
139
0
  UnlockReleaseBuffer(buffer);
140
141
0
  return freesize;
142
0
}
143
144
static void
145
makeSublist(Relation index, IndexTuple *tuples, int32 ntuples,
146
      GinMetaPageData *res)
147
0
{
148
0
  Buffer    curBuffer = InvalidBuffer;
149
0
  Buffer    prevBuffer = InvalidBuffer;
150
0
  int     i,
151
0
        size = 0,
152
0
        tupsize;
153
0
  int     startTuple = 0;
154
155
0
  Assert(ntuples > 0);
156
157
  /*
158
   * Split tuples into pages
159
   */
160
0
  for (i = 0; i < ntuples; i++)
161
0
  {
162
0
    if (curBuffer == InvalidBuffer)
163
0
    {
164
0
      curBuffer = GinNewBuffer(index);
165
166
0
      if (prevBuffer != InvalidBuffer)
167
0
      {
168
0
        res->nPendingPages++;
169
0
        writeListPage(index, prevBuffer,
170
0
                tuples + startTuple,
171
0
                i - startTuple,
172
0
                BufferGetBlockNumber(curBuffer));
173
0
      }
174
0
      else
175
0
      {
176
0
        res->head = BufferGetBlockNumber(curBuffer);
177
0
      }
178
179
0
      prevBuffer = curBuffer;
180
0
      startTuple = i;
181
0
      size = 0;
182
0
    }
183
184
0
    tupsize = MAXALIGN(IndexTupleSize(tuples[i])) + sizeof(ItemIdData);
185
186
0
    if (size + tupsize > GinListPageSize)
187
0
    {
188
      /* won't fit, force a new page and reprocess */
189
0
      i--;
190
0
      curBuffer = InvalidBuffer;
191
0
    }
192
0
    else
193
0
    {
194
0
      size += tupsize;
195
0
    }
196
0
  }
197
198
  /*
199
   * Write last page
200
   */
201
0
  res->tail = BufferGetBlockNumber(curBuffer);
202
0
  res->tailFreeSize = writeListPage(index, curBuffer,
203
0
                    tuples + startTuple,
204
0
                    ntuples - startTuple,
205
0
                    InvalidBlockNumber);
206
0
  res->nPendingPages++;
207
  /* that was only one heap tuple */
208
0
  res->nPendingHeapTuples = 1;
209
0
}
210
211
/*
212
 * Write the index tuples contained in *collector into the index's
213
 * pending list.
214
 *
215
 * Function guarantees that all these tuples will be inserted consecutively,
216
 * preserving order
217
 */
218
void
219
ginHeapTupleFastInsert(GinState *ginstate, GinTupleCollector *collector)
220
0
{
221
0
  Relation  index = ginstate->index;
222
0
  Buffer    metabuffer;
223
0
  Page    metapage;
224
0
  GinMetaPageData *metadata = NULL;
225
0
  Buffer    buffer = InvalidBuffer;
226
0
  Page    page = NULL;
227
0
  ginxlogUpdateMeta data;
228
0
  bool    separateList = false;
229
0
  bool    needCleanup = false;
230
0
  int     cleanupSize;
231
0
  bool    needWal;
232
233
0
  if (collector->ntuples == 0)
234
0
    return;
235
236
0
  needWal = RelationNeedsWAL(index);
237
238
0
  data.locator = index->rd_locator;
239
0
  data.ntuples = 0;
240
0
  data.newRightlink = data.prevTail = InvalidBlockNumber;
241
242
0
  metabuffer = ReadBuffer(index, GIN_METAPAGE_BLKNO);
243
0
  metapage = BufferGetPage(metabuffer);
244
245
  /*
246
   * An insertion to the pending list could logically belong anywhere in the
247
   * tree, so it conflicts with all serializable scans.  All scans acquire a
248
   * predicate lock on the metabuffer to represent that.  Therefore we'll
249
   * check for conflicts in, but not until we have the page locked and are
250
   * ready to modify the page.
251
   */
252
253
0
  if (collector->sumsize + collector->ntuples * sizeof(ItemIdData) > GinListPageSize)
254
0
  {
255
    /*
256
     * Total size is greater than one page => make sublist
257
     */
258
0
    separateList = true;
259
0
  }
260
0
  else
261
0
  {
262
0
    LockBuffer(metabuffer, GIN_EXCLUSIVE);
263
0
    metadata = GinPageGetMeta(metapage);
264
265
0
    if (metadata->head == InvalidBlockNumber ||
266
0
      collector->sumsize + collector->ntuples * sizeof(ItemIdData) > metadata->tailFreeSize)
267
0
    {
268
      /*
269
       * Pending list is empty or total size is greater than freespace
270
       * on tail page => make sublist
271
       *
272
       * We unlock metabuffer to keep high concurrency
273
       */
274
0
      separateList = true;
275
0
      LockBuffer(metabuffer, GIN_UNLOCK);
276
0
    }
277
0
  }
278
279
0
  if (separateList)
280
0
  {
281
    /*
282
     * We should make sublist separately and append it to the tail
283
     */
284
0
    GinMetaPageData sublist;
285
286
0
    memset(&sublist, 0, sizeof(GinMetaPageData));
287
0
    makeSublist(index, collector->tuples, collector->ntuples, &sublist);
288
289
    /*
290
     * metapage was unlocked, see above
291
     */
292
0
    LockBuffer(metabuffer, GIN_EXCLUSIVE);
293
0
    metadata = GinPageGetMeta(metapage);
294
295
0
    CheckForSerializableConflictIn(index, NULL, GIN_METAPAGE_BLKNO);
296
297
0
    if (metadata->head == InvalidBlockNumber)
298
0
    {
299
      /*
300
       * Main list is empty, so just insert sublist as main list
301
       */
302
0
      START_CRIT_SECTION();
303
304
0
      metadata->head = sublist.head;
305
0
      metadata->tail = sublist.tail;
306
0
      metadata->tailFreeSize = sublist.tailFreeSize;
307
308
0
      metadata->nPendingPages = sublist.nPendingPages;
309
0
      metadata->nPendingHeapTuples = sublist.nPendingHeapTuples;
310
311
0
      if (needWal)
312
0
        XLogBeginInsert();
313
0
    }
314
0
    else
315
0
    {
316
      /*
317
       * Merge lists
318
       */
319
0
      data.prevTail = metadata->tail;
320
0
      data.newRightlink = sublist.head;
321
322
0
      buffer = ReadBuffer(index, metadata->tail);
323
0
      LockBuffer(buffer, GIN_EXCLUSIVE);
324
0
      page = BufferGetPage(buffer);
325
326
0
      Assert(GinPageGetOpaque(page)->rightlink == InvalidBlockNumber);
327
328
0
      START_CRIT_SECTION();
329
330
0
      GinPageGetOpaque(page)->rightlink = sublist.head;
331
332
0
      MarkBufferDirty(buffer);
333
334
0
      metadata->tail = sublist.tail;
335
0
      metadata->tailFreeSize = sublist.tailFreeSize;
336
337
0
      metadata->nPendingPages += sublist.nPendingPages;
338
0
      metadata->nPendingHeapTuples += sublist.nPendingHeapTuples;
339
340
0
      if (needWal)
341
0
      {
342
0
        XLogBeginInsert();
343
0
        XLogRegisterBuffer(1, buffer, REGBUF_STANDARD);
344
0
      }
345
0
    }
346
0
  }
347
0
  else
348
0
  {
349
    /*
350
     * Insert into tail page.  Metapage is already locked
351
     */
352
0
    OffsetNumber l,
353
0
          off;
354
0
    int     i,
355
0
          tupsize;
356
0
    char     *ptr;
357
0
    char     *collectordata;
358
359
0
    CheckForSerializableConflictIn(index, NULL, GIN_METAPAGE_BLKNO);
360
361
0
    buffer = ReadBuffer(index, metadata->tail);
362
0
    LockBuffer(buffer, GIN_EXCLUSIVE);
363
0
    page = BufferGetPage(buffer);
364
365
0
    off = (PageIsEmpty(page)) ? FirstOffsetNumber :
366
0
      OffsetNumberNext(PageGetMaxOffsetNumber(page));
367
368
0
    collectordata = ptr = (char *) palloc(collector->sumsize);
369
370
0
    data.ntuples = collector->ntuples;
371
372
0
    START_CRIT_SECTION();
373
374
0
    if (needWal)
375
0
      XLogBeginInsert();
376
377
    /*
378
     * Increase counter of heap tuples
379
     */
380
0
    Assert(GinPageGetOpaque(page)->maxoff <= metadata->nPendingHeapTuples);
381
0
    GinPageGetOpaque(page)->maxoff++;
382
0
    metadata->nPendingHeapTuples++;
383
384
0
    for (i = 0; i < collector->ntuples; i++)
385
0
    {
386
0
      tupsize = IndexTupleSize(collector->tuples[i]);
387
0
      l = PageAddItem(page, collector->tuples[i], tupsize, off, false, false);
388
389
0
      if (l == InvalidOffsetNumber)
390
0
        elog(ERROR, "failed to add item to index page in \"%s\"",
391
0
           RelationGetRelationName(index));
392
393
0
      memcpy(ptr, collector->tuples[i], tupsize);
394
0
      ptr += tupsize;
395
396
0
      off++;
397
0
    }
398
399
0
    Assert((ptr - collectordata) <= collector->sumsize);
400
401
0
    MarkBufferDirty(buffer);
402
403
0
    if (needWal)
404
0
    {
405
0
      XLogRegisterBuffer(1, buffer, REGBUF_STANDARD);
406
0
      XLogRegisterBufData(1, collectordata, collector->sumsize);
407
0
    }
408
409
0
    metadata->tailFreeSize = PageGetExactFreeSpace(page);
410
0
  }
411
412
  /*
413
   * Set pd_lower just past the end of the metadata.  This is essential,
414
   * because without doing so, metadata will be lost if xlog.c compresses
415
   * the page.  (We must do this here because pre-v11 versions of PG did not
416
   * set the metapage's pd_lower correctly, so a pg_upgraded index might
417
   * contain the wrong value.)
418
   */
419
0
  ((PageHeader) metapage)->pd_lower =
420
0
    ((char *) metadata + sizeof(GinMetaPageData)) - (char *) metapage;
421
422
  /*
423
   * Write metabuffer, make xlog entry
424
   */
425
0
  MarkBufferDirty(metabuffer);
426
427
0
  if (needWal)
428
0
  {
429
0
    XLogRecPtr  recptr;
430
431
0
    memcpy(&data.metadata, metadata, sizeof(GinMetaPageData));
432
433
0
    XLogRegisterBuffer(0, metabuffer, REGBUF_WILL_INIT | REGBUF_STANDARD);
434
0
    XLogRegisterData(&data, sizeof(ginxlogUpdateMeta));
435
436
0
    recptr = XLogInsert(RM_GIN_ID, XLOG_GIN_UPDATE_META_PAGE);
437
0
    PageSetLSN(metapage, recptr);
438
439
0
    if (buffer != InvalidBuffer)
440
0
    {
441
0
      PageSetLSN(page, recptr);
442
0
    }
443
0
  }
444
445
0
  if (buffer != InvalidBuffer)
446
0
    UnlockReleaseBuffer(buffer);
447
448
  /*
449
   * Force pending list cleanup when it becomes too long. And,
450
   * ginInsertCleanup could take significant amount of time, so we prefer to
451
   * call it when it can do all the work in a single collection cycle. In
452
   * non-vacuum mode, it shouldn't require maintenance_work_mem, so fire it
453
   * while pending list is still small enough to fit into
454
   * gin_pending_list_limit.
455
   *
456
   * ginInsertCleanup() should not be called inside our CRIT_SECTION.
457
   */
458
0
  cleanupSize = GinGetPendingListCleanupSize(index);
459
0
  if (metadata->nPendingPages * GIN_PAGE_FREESIZE > cleanupSize * (Size) 1024)
460
0
    needCleanup = true;
461
462
0
  END_CRIT_SECTION();
463
464
0
  UnlockReleaseBuffer(metabuffer);
465
466
  /*
467
   * Since it could contend with concurrent cleanup process we cleanup
468
   * pending list not forcibly.
469
   */
470
0
  if (needCleanup)
471
0
    ginInsertCleanup(ginstate, false, true, false, NULL);
472
0
}
473
474
/*
475
 * Create temporary index tuples for a single indexable item (one index column
476
 * for the heap tuple specified by ht_ctid), and append them to the array
477
 * in *collector.  They will subsequently be written out using
478
 * ginHeapTupleFastInsert.  Note that to guarantee consistent state, all
479
 * temp tuples for a given heap tuple must be written in one call to
480
 * ginHeapTupleFastInsert.
481
 */
482
void
483
ginHeapTupleFastCollect(GinState *ginstate,
484
            GinTupleCollector *collector,
485
            OffsetNumber attnum, Datum value, bool isNull,
486
            ItemPointer ht_ctid)
487
0
{
488
0
  Datum    *entries;
489
0
  GinNullCategory *categories;
490
0
  int32   i,
491
0
        nentries;
492
493
  /*
494
   * Extract the key values that need to be inserted in the index
495
   */
496
0
  entries = ginExtractEntries(ginstate, attnum, value, isNull,
497
0
                &nentries, &categories);
498
499
  /*
500
   * Protect against integer overflow in allocation calculations
501
   */
502
0
  if (nentries < 0 ||
503
0
    collector->ntuples + nentries > MaxAllocSize / sizeof(IndexTuple))
504
0
    elog(ERROR, "too many entries for GIN index");
505
506
  /*
507
   * Allocate/reallocate memory for storing collected tuples
508
   */
509
0
  if (collector->tuples == NULL)
510
0
  {
511
    /*
512
     * Determine the number of elements to allocate in the tuples array
513
     * initially.  Make it a power of 2 to avoid wasting memory when
514
     * resizing (since palloc likes powers of 2).
515
     */
516
0
    collector->lentuples = pg_nextpower2_32(Max(16, nentries));
517
0
    collector->tuples = palloc_array(IndexTuple, collector->lentuples);
518
0
  }
519
0
  else if (collector->lentuples < collector->ntuples + nentries)
520
0
  {
521
    /*
522
     * Advance lentuples to the next suitable power of 2.  This won't
523
     * overflow, though we could get to a value that exceeds
524
     * MaxAllocSize/sizeof(IndexTuple), causing an error in repalloc.
525
     */
526
0
    collector->lentuples = pg_nextpower2_32(collector->ntuples + nentries);
527
0
    collector->tuples = repalloc_array(collector->tuples,
528
0
                       IndexTuple, collector->lentuples);
529
0
  }
530
531
  /*
532
   * Build an index tuple for each key value, and add to array.  In pending
533
   * tuples we just stick the heap TID into t_tid.
534
   */
535
0
  for (i = 0; i < nentries; i++)
536
0
  {
537
0
    IndexTuple  itup;
538
539
0
    itup = GinFormTuple(ginstate, attnum, entries[i], categories[i],
540
0
              NULL, 0, 0, true);
541
0
    itup->t_tid = *ht_ctid;
542
0
    collector->tuples[collector->ntuples++] = itup;
543
0
    collector->sumsize += IndexTupleSize(itup);
544
0
  }
545
0
}
546
547
/*
548
 * Deletes pending list pages up to (not including) newHead page.
549
 * If newHead == InvalidBlockNumber then function drops the whole list.
550
 *
551
 * metapage is pinned and exclusive-locked throughout this function.
552
 */
553
static void
554
shiftList(Relation index, Buffer metabuffer, BlockNumber newHead,
555
      bool fill_fsm, IndexBulkDeleteResult *stats)
556
0
{
557
0
  Page    metapage;
558
0
  GinMetaPageData *metadata;
559
0
  BlockNumber blknoToDelete;
560
561
0
  metapage = BufferGetPage(metabuffer);
562
0
  metadata = GinPageGetMeta(metapage);
563
0
  blknoToDelete = metadata->head;
564
565
0
  do
566
0
  {
567
0
    Page    page;
568
0
    int     i;
569
0
    int64   nDeletedHeapTuples = 0;
570
0
    ginxlogDeleteListPages data;
571
0
    Buffer    buffers[GIN_NDELETE_AT_ONCE];
572
0
    BlockNumber freespace[GIN_NDELETE_AT_ONCE];
573
574
0
    data.ndeleted = 0;
575
0
    while (data.ndeleted < GIN_NDELETE_AT_ONCE && blknoToDelete != newHead)
576
0
    {
577
0
      freespace[data.ndeleted] = blknoToDelete;
578
0
      buffers[data.ndeleted] = ReadBuffer(index, blknoToDelete);
579
0
      LockBuffer(buffers[data.ndeleted], GIN_EXCLUSIVE);
580
0
      page = BufferGetPage(buffers[data.ndeleted]);
581
582
0
      data.ndeleted++;
583
584
0
      Assert(!GinPageIsDeleted(page));
585
586
0
      nDeletedHeapTuples += GinPageGetOpaque(page)->maxoff;
587
0
      blknoToDelete = GinPageGetOpaque(page)->rightlink;
588
0
    }
589
590
0
    if (stats)
591
0
      stats->pages_deleted += data.ndeleted;
592
593
    /*
594
     * This operation touches an unusually large number of pages, so
595
     * prepare the XLogInsert machinery for that before entering the
596
     * critical section.
597
     */
598
0
    if (RelationNeedsWAL(index))
599
0
      XLogEnsureRecordSpace(data.ndeleted, 0);
600
601
0
    START_CRIT_SECTION();
602
603
0
    metadata->head = blknoToDelete;
604
605
0
    Assert(metadata->nPendingPages >= data.ndeleted);
606
0
    metadata->nPendingPages -= data.ndeleted;
607
0
    Assert(metadata->nPendingHeapTuples >= nDeletedHeapTuples);
608
0
    metadata->nPendingHeapTuples -= nDeletedHeapTuples;
609
610
0
    if (blknoToDelete == InvalidBlockNumber)
611
0
    {
612
0
      metadata->tail = InvalidBlockNumber;
613
0
      metadata->tailFreeSize = 0;
614
0
      metadata->nPendingPages = 0;
615
0
      metadata->nPendingHeapTuples = 0;
616
0
    }
617
618
    /*
619
     * Set pd_lower just past the end of the metadata.  This is essential,
620
     * because without doing so, metadata will be lost if xlog.c
621
     * compresses the page.  (We must do this here because pre-v11
622
     * versions of PG did not set the metapage's pd_lower correctly, so a
623
     * pg_upgraded index might contain the wrong value.)
624
     */
625
0
    ((PageHeader) metapage)->pd_lower =
626
0
      ((char *) metadata + sizeof(GinMetaPageData)) - (char *) metapage;
627
628
0
    MarkBufferDirty(metabuffer);
629
630
0
    for (i = 0; i < data.ndeleted; i++)
631
0
    {
632
0
      page = BufferGetPage(buffers[i]);
633
0
      GinPageGetOpaque(page)->flags = GIN_DELETED;
634
0
      MarkBufferDirty(buffers[i]);
635
0
    }
636
637
0
    if (RelationNeedsWAL(index))
638
0
    {
639
0
      XLogRecPtr  recptr;
640
641
0
      XLogBeginInsert();
642
0
      XLogRegisterBuffer(0, metabuffer,
643
0
                 REGBUF_WILL_INIT | REGBUF_STANDARD);
644
0
      for (i = 0; i < data.ndeleted; i++)
645
0
        XLogRegisterBuffer(i + 1, buffers[i], REGBUF_WILL_INIT);
646
647
0
      memcpy(&data.metadata, metadata, sizeof(GinMetaPageData));
648
649
0
      XLogRegisterData(&data,
650
0
               sizeof(ginxlogDeleteListPages));
651
652
0
      recptr = XLogInsert(RM_GIN_ID, XLOG_GIN_DELETE_LISTPAGE);
653
0
      PageSetLSN(metapage, recptr);
654
655
0
      for (i = 0; i < data.ndeleted; i++)
656
0
      {
657
0
        page = BufferGetPage(buffers[i]);
658
0
        PageSetLSN(page, recptr);
659
0
      }
660
0
    }
661
662
0
    END_CRIT_SECTION();
663
664
0
    for (i = 0; i < data.ndeleted; i++)
665
0
      UnlockReleaseBuffer(buffers[i]);
666
667
0
    for (i = 0; fill_fsm && i < data.ndeleted; i++)
668
0
      RecordFreeIndexPage(index, freespace[i]);
669
670
0
  } while (blknoToDelete != newHead);
671
0
}
672
673
/* Initialize empty KeyArray */
674
static void
675
initKeyArray(KeyArray *keys, int32 maxvalues)
676
0
{
677
0
  keys->keys = palloc_array(Datum, maxvalues);
678
0
  keys->categories = palloc_array(GinNullCategory, maxvalues);
679
0
  keys->nvalues = 0;
680
0
  keys->maxvalues = maxvalues;
681
0
}
682
683
/* Add datum to KeyArray, resizing if needed */
684
static void
685
addDatum(KeyArray *keys, Datum datum, GinNullCategory category)
686
0
{
687
0
  if (keys->nvalues >= keys->maxvalues)
688
0
  {
689
0
    keys->maxvalues *= 2;
690
0
    keys->keys = repalloc_array(keys->keys, Datum, keys->maxvalues);
691
0
    keys->categories = repalloc_array(keys->categories, GinNullCategory, keys->maxvalues);
692
0
  }
693
694
0
  keys->keys[keys->nvalues] = datum;
695
0
  keys->categories[keys->nvalues] = category;
696
0
  keys->nvalues++;
697
0
}
698
699
/*
700
 * Collect data from a pending-list page in preparation for insertion into
701
 * the main index.
702
 *
703
 * Go through all tuples >= startoff on page and collect values in accum
704
 *
705
 * Note that ka is just workspace --- it does not carry any state across
706
 * calls.
707
 */
708
static void
709
processPendingPage(BuildAccumulator *accum, KeyArray *ka,
710
           Page page, OffsetNumber startoff)
711
0
{
712
0
  ItemPointerData heapptr;
713
0
  OffsetNumber i,
714
0
        maxoff;
715
0
  OffsetNumber attrnum;
716
717
  /* reset *ka to empty */
718
0
  ka->nvalues = 0;
719
720
0
  maxoff = PageGetMaxOffsetNumber(page);
721
0
  Assert(maxoff >= FirstOffsetNumber);
722
0
  ItemPointerSetInvalid(&heapptr);
723
0
  attrnum = 0;
724
725
0
  for (i = startoff; i <= maxoff; i = OffsetNumberNext(i))
726
0
  {
727
0
    IndexTuple  itup = (IndexTuple) PageGetItem(page, PageGetItemId(page, i));
728
0
    OffsetNumber curattnum;
729
0
    Datum   curkey;
730
0
    GinNullCategory curcategory;
731
732
    /* Check for change of heap TID or attnum */
733
0
    curattnum = gintuple_get_attrnum(accum->ginstate, itup);
734
735
0
    if (!ItemPointerIsValid(&heapptr))
736
0
    {
737
0
      heapptr = itup->t_tid;
738
0
      attrnum = curattnum;
739
0
    }
740
0
    else if (!(ItemPointerEquals(&heapptr, &itup->t_tid) &&
741
0
           curattnum == attrnum))
742
0
    {
743
      /*
744
       * ginInsertBAEntries can insert several datums per call, but only
745
       * for one heap tuple and one column.  So call it at a boundary,
746
       * and reset ka.
747
       */
748
0
      ginInsertBAEntries(accum, &heapptr, attrnum,
749
0
                 ka->keys, ka->categories, ka->nvalues);
750
0
      ka->nvalues = 0;
751
0
      heapptr = itup->t_tid;
752
0
      attrnum = curattnum;
753
0
    }
754
755
    /* Add key to KeyArray */
756
0
    curkey = gintuple_get_key(accum->ginstate, itup, &curcategory);
757
0
    addDatum(ka, curkey, curcategory);
758
0
  }
759
760
  /* Dump out all remaining keys */
761
0
  ginInsertBAEntries(accum, &heapptr, attrnum,
762
0
             ka->keys, ka->categories, ka->nvalues);
763
0
}
764
765
/*
766
 * Move tuples from pending pages into regular GIN structure.
767
 *
768
 * On first glance it looks completely not crash-safe. But if we crash
769
 * after posting entries to the main index and before removing them from the
770
 * pending list, it's okay because when we redo the posting later on, nothing
771
 * bad will happen.
772
 *
773
 * fill_fsm indicates that ginInsertCleanup should add deleted pages
774
 * to FSM otherwise caller is responsible to put deleted pages into
775
 * FSM.
776
 *
777
 * If stats isn't null, we count deleted pending pages into the counts.
778
 */
779
void
780
ginInsertCleanup(GinState *ginstate, bool full_clean,
781
         bool fill_fsm, bool forceCleanup,
782
         IndexBulkDeleteResult *stats)
783
0
{
784
0
  Relation  index = ginstate->index;
785
0
  Buffer    metabuffer,
786
0
        buffer;
787
0
  Page    metapage,
788
0
        page;
789
0
  GinMetaPageData *metadata;
790
0
  MemoryContext opCtx,
791
0
        oldCtx;
792
0
  BuildAccumulator accum;
793
0
  KeyArray  datums;
794
0
  BlockNumber blkno,
795
0
        blknoFinish;
796
0
  bool    cleanupFinish = false;
797
0
  bool    fsm_vac = false;
798
0
  int     workMemory;
799
800
  /*
801
   * We would like to prevent concurrent cleanup process. For that we will
802
   * lock metapage in exclusive mode using LockPage() call. Nobody other
803
   * will use that lock for metapage, so we keep possibility of concurrent
804
   * insertion into pending list
805
   */
806
807
0
  if (forceCleanup)
808
0
  {
809
    /*
810
     * We are called from [auto]vacuum/analyze or gin_clean_pending_list()
811
     * and we would like to wait concurrent cleanup to finish.
812
     */
813
0
    LockPage(index, GIN_METAPAGE_BLKNO, ExclusiveLock);
814
0
    workMemory =
815
0
      (AmAutoVacuumWorkerProcess() && autovacuum_work_mem != -1) ?
816
0
      autovacuum_work_mem : maintenance_work_mem;
817
0
  }
818
0
  else
819
0
  {
820
    /*
821
     * We are called from regular insert and if we see concurrent cleanup
822
     * just exit in hope that concurrent process will clean up pending
823
     * list.
824
     */
825
0
    if (!ConditionalLockPage(index, GIN_METAPAGE_BLKNO, ExclusiveLock))
826
0
      return;
827
0
    workMemory = work_mem;
828
0
  }
829
830
0
  metabuffer = ReadBuffer(index, GIN_METAPAGE_BLKNO);
831
0
  LockBuffer(metabuffer, GIN_SHARE);
832
0
  metapage = BufferGetPage(metabuffer);
833
0
  metadata = GinPageGetMeta(metapage);
834
835
0
  if (metadata->head == InvalidBlockNumber)
836
0
  {
837
    /* Nothing to do */
838
0
    UnlockReleaseBuffer(metabuffer);
839
0
    UnlockPage(index, GIN_METAPAGE_BLKNO, ExclusiveLock);
840
0
    return;
841
0
  }
842
843
  /*
844
   * Remember a tail page to prevent infinite cleanup if other backends add
845
   * new tuples faster than we can cleanup.
846
   */
847
0
  blknoFinish = metadata->tail;
848
849
  /*
850
   * Read and lock head of pending list
851
   */
852
0
  blkno = metadata->head;
853
0
  buffer = ReadBuffer(index, blkno);
854
0
  LockBuffer(buffer, GIN_SHARE);
855
0
  page = BufferGetPage(buffer);
856
857
0
  LockBuffer(metabuffer, GIN_UNLOCK);
858
859
  /*
860
   * Initialize.  All temporary space will be in opCtx
861
   */
862
0
  opCtx = AllocSetContextCreate(CurrentMemoryContext,
863
0
                  "GIN insert cleanup temporary context",
864
0
                  ALLOCSET_DEFAULT_SIZES);
865
866
0
  oldCtx = MemoryContextSwitchTo(opCtx);
867
868
0
  initKeyArray(&datums, 128);
869
0
  ginInitBA(&accum);
870
0
  accum.ginstate = ginstate;
871
872
  /*
873
   * At the top of this loop, we have pin and lock on the current page of
874
   * the pending list.  However, we'll release that before exiting the loop.
875
   * Note we also have pin but not lock on the metapage.
876
   */
877
0
  for (;;)
878
0
  {
879
0
    Assert(!GinPageIsDeleted(page));
880
881
    /*
882
     * Are we walk through the page which as we remember was a tail when
883
     * we start our cleanup?  But if caller asks us to clean up whole
884
     * pending list then ignore old tail, we will work until list becomes
885
     * empty.
886
     */
887
0
    if (blkno == blknoFinish && full_clean == false)
888
0
      cleanupFinish = true;
889
890
    /*
891
     * read page's datums into accum
892
     */
893
0
    processPendingPage(&accum, &datums, page, FirstOffsetNumber);
894
895
0
    vacuum_delay_point(false);
896
897
    /*
898
     * Is it time to flush memory to disk?  Flush if we are at the end of
899
     * the pending list, or if we have a full row and memory is getting
900
     * full.
901
     */
902
0
    if (GinPageGetOpaque(page)->rightlink == InvalidBlockNumber ||
903
0
      (GinPageHasFullRow(page) &&
904
0
       accum.allocatedMemory >= workMemory * (Size) 1024))
905
0
    {
906
0
      ItemPointerData *list;
907
0
      uint32    nlist;
908
0
      Datum   key;
909
0
      GinNullCategory category;
910
0
      OffsetNumber maxoff,
911
0
            attnum;
912
913
      /*
914
       * Unlock current page to increase performance. Changes of page
915
       * will be checked later by comparing maxoff after completion of
916
       * memory flush.
917
       */
918
0
      maxoff = PageGetMaxOffsetNumber(page);
919
0
      LockBuffer(buffer, GIN_UNLOCK);
920
921
      /*
922
       * Moving collected data into regular structure can take
923
       * significant amount of time - so, run it without locking pending
924
       * list.
925
       */
926
0
      ginBeginBAScan(&accum);
927
0
      while ((list = ginGetBAEntry(&accum,
928
0
                     &attnum, &key, &category, &nlist)) != NULL)
929
0
      {
930
0
        ginEntryInsert(ginstate, attnum, key, category,
931
0
                 list, nlist, NULL);
932
0
        vacuum_delay_point(false);
933
0
      }
934
935
      /*
936
       * Lock the whole list to remove pages
937
       */
938
0
      LockBuffer(metabuffer, GIN_EXCLUSIVE);
939
0
      LockBuffer(buffer, GIN_SHARE);
940
941
0
      Assert(!GinPageIsDeleted(page));
942
943
      /*
944
       * While we left the page unlocked, more stuff might have gotten
945
       * added to it.  If so, process those entries immediately.  There
946
       * shouldn't be very many, so we don't worry about the fact that
947
       * we're doing this with exclusive lock. Insertion algorithm
948
       * guarantees that inserted row(s) will not continue on next page.
949
       * NOTE: intentionally no vacuum_delay_point in this loop.
950
       */
951
0
      if (PageGetMaxOffsetNumber(page) != maxoff)
952
0
      {
953
0
        ginInitBA(&accum);
954
0
        processPendingPage(&accum, &datums, page, maxoff + 1);
955
956
0
        ginBeginBAScan(&accum);
957
0
        while ((list = ginGetBAEntry(&accum,
958
0
                       &attnum, &key, &category, &nlist)) != NULL)
959
0
          ginEntryInsert(ginstate, attnum, key, category,
960
0
                   list, nlist, NULL);
961
0
      }
962
963
      /*
964
       * Remember next page - it will become the new list head
965
       */
966
0
      blkno = GinPageGetOpaque(page)->rightlink;
967
0
      UnlockReleaseBuffer(buffer);  /* shiftList will do exclusive
968
                       * locking */
969
970
      /*
971
       * remove read pages from pending list, at this point all content
972
       * of read pages is in regular structure
973
       */
974
0
      shiftList(index, metabuffer, blkno, fill_fsm, stats);
975
976
      /* At this point, some pending pages have been freed up */
977
0
      fsm_vac = true;
978
979
0
      Assert(blkno == metadata->head);
980
0
      LockBuffer(metabuffer, GIN_UNLOCK);
981
982
      /*
983
       * if we removed the whole pending list or we cleanup tail (which
984
       * we remembered on start our cleanup process) then just exit
985
       */
986
0
      if (blkno == InvalidBlockNumber || cleanupFinish)
987
0
        break;
988
989
      /*
990
       * release memory used so far and reinit state
991
       */
992
0
      MemoryContextReset(opCtx);
993
0
      initKeyArray(&datums, datums.maxvalues);
994
0
      ginInitBA(&accum);
995
0
    }
996
0
    else
997
0
    {
998
0
      blkno = GinPageGetOpaque(page)->rightlink;
999
0
      UnlockReleaseBuffer(buffer);
1000
0
    }
1001
1002
    /*
1003
     * Read next page in pending list
1004
     */
1005
0
    vacuum_delay_point(false);
1006
0
    buffer = ReadBuffer(index, blkno);
1007
0
    LockBuffer(buffer, GIN_SHARE);
1008
0
    page = BufferGetPage(buffer);
1009
0
  }
1010
1011
0
  UnlockPage(index, GIN_METAPAGE_BLKNO, ExclusiveLock);
1012
0
  ReleaseBuffer(metabuffer);
1013
1014
  /*
1015
   * As pending list pages can have a high churn rate, it is desirable to
1016
   * recycle them immediately to the FreeSpaceMap when ordinary backends
1017
   * clean the list.
1018
   */
1019
0
  if (fsm_vac && fill_fsm)
1020
0
    IndexFreeSpaceMapVacuum(index);
1021
1022
  /* Clean up temporary space */
1023
0
  MemoryContextSwitchTo(oldCtx);
1024
0
  MemoryContextDelete(opCtx);
1025
0
}
1026
1027
/*
1028
 * SQL-callable function to clean the insert pending list
1029
 */
1030
Datum
1031
gin_clean_pending_list(PG_FUNCTION_ARGS)
1032
{
1033
  Oid     indexoid = PG_GETARG_OID(0);
1034
  Relation  indexRel = index_open(indexoid, RowExclusiveLock);
1035
  IndexBulkDeleteResult stats;
1036
1037
  if (RecoveryInProgress())
1038
    ereport(ERROR,
1039
        (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE),
1040
         errmsg("recovery is in progress"),
1041
         errhint("GIN pending list cannot be cleaned up during recovery.")));
1042
1043
  /* Must be a GIN index */
1044
  if (indexRel->rd_rel->relkind != RELKIND_INDEX ||
1045
    indexRel->rd_rel->relam != GIN_AM_OID)
1046
    ereport(ERROR,
1047
        (errcode(ERRCODE_WRONG_OBJECT_TYPE),
1048
         errmsg("\"%s\" is not a GIN index",
1049
            RelationGetRelationName(indexRel))));
1050
1051
  /*
1052
   * Reject attempts to read non-local temporary relations; we would be
1053
   * likely to get wrong data since we have no visibility into the owning
1054
   * session's local buffers.
1055
   */
1056
  if (RELATION_IS_OTHER_TEMP(indexRel))
1057
    ereport(ERROR,
1058
        (errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
1059
         errmsg("cannot access temporary indexes of other sessions")));
1060
1061
  /* User must own the index (comparable to privileges needed for VACUUM) */
1062
  if (!object_ownercheck(RelationRelationId, indexoid, GetUserId()))
1063
    aclcheck_error(ACLCHECK_NOT_OWNER, OBJECT_INDEX,
1064
             RelationGetRelationName(indexRel));
1065
1066
  memset(&stats, 0, sizeof(stats));
1067
1068
  /*
1069
   * Can't assume anything about the content of an !indisready index.  Make
1070
   * those a no-op, not an error, so users can just run this function on all
1071
   * indexes of the access method.  Since an indisready&&!indisvalid index
1072
   * is merely awaiting missed aminsert calls, we're capable of processing
1073
   * it.  Decline to do so, out of an abundance of caution.
1074
   */
1075
  if (indexRel->rd_index->indisvalid)
1076
  {
1077
    GinState  ginstate;
1078
1079
    initGinState(&ginstate, indexRel);
1080
    ginInsertCleanup(&ginstate, true, true, true, &stats);
1081
  }
1082
  else
1083
    ereport(DEBUG1,
1084
        (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE),
1085
         errmsg("index \"%s\" is not valid",
1086
            RelationGetRelationName(indexRel))));
1087
1088
  index_close(indexRel, RowExclusiveLock);
1089
1090
  PG_RETURN_INT64((int64) stats.pages_deleted);
1091
}