Coverage Report

Created: 2026-09-28 06:55

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/postgres/src/backend/utils/adt/tsvector.c
Line
Count
Source
1
/*-------------------------------------------------------------------------
2
 *
3
 * tsvector.c
4
 *    I/O functions for tsvector
5
 *
6
 * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
7
 *
8
 *
9
 * IDENTIFICATION
10
 *    src/backend/utils/adt/tsvector.c
11
 *
12
 *-------------------------------------------------------------------------
13
 */
14
15
#include "postgres.h"
16
17
#include "common/int.h"
18
#include "libpq/pqformat.h"
19
#include "nodes/miscnodes.h"
20
#include "tsearch/ts_locale.h"
21
#include "tsearch/ts_utils.h"
22
#include "utils/fmgrprotos.h"
23
#include "utils/memutils.h"
24
#include "varatt.h"
25
26
typedef struct
27
{
28
  WordEntry entry;      /* must be first, see compareentry */
29
  WordEntryPos *pos;
30
  int     poslen;     /* number of elements in pos */
31
} WordEntryIN;
32
33
34
/* Compare two WordEntryPos values for qsort */
35
int
36
compareWordEntryPos(const void *a, const void *b)
37
0
{
38
0
  int     apos = WEP_GETPOS(*(const WordEntryPos *) a);
39
0
  int     bpos = WEP_GETPOS(*(const WordEntryPos *) b);
40
41
0
  return pg_cmp_s32(apos, bpos);
42
0
}
43
44
/*
45
 * Removes duplicate pos entries. If there's two entries with same pos but
46
 * different weight, the higher weight is retained, so we can't use
47
 * qunique here.
48
 *
49
 * Returns new length.
50
 */
51
static int
52
uniquePos(WordEntryPos *a, int l)
53
0
{
54
0
  WordEntryPos *ptr,
55
0
         *res;
56
57
0
  if (l <= 1)
58
0
    return l;
59
60
0
  qsort(a, l, sizeof(WordEntryPos), compareWordEntryPos);
61
62
0
  res = a;
63
0
  ptr = a + 1;
64
0
  while (ptr - a < l)
65
0
  {
66
0
    if (WEP_GETPOS(*ptr) != WEP_GETPOS(*res))
67
0
    {
68
0
      res++;
69
0
      *res = *ptr;
70
0
      if (res - a >= MAXNUMPOS - 1 ||
71
0
        WEP_GETPOS(*res) == MAXENTRYPOS - 1)
72
0
        break;
73
0
    }
74
0
    else if (WEP_GETWEIGHT(*ptr) > WEP_GETWEIGHT(*res))
75
0
      WEP_SETWEIGHT(*res, WEP_GETWEIGHT(*ptr));
76
0
    ptr++;
77
0
  }
78
79
0
  return res + 1 - a;
80
0
}
81
82
/*
83
 * Compare two WordEntry structs for qsort_arg.  This can also be used on
84
 * WordEntryIN structs, since those have WordEntry as their first field.
85
 */
86
static int
87
compareentry(const void *va, const void *vb, void *arg)
88
0
{
89
0
  const WordEntry *a = (const WordEntry *) va;
90
0
  const WordEntry *b = (const WordEntry *) vb;
91
0
  char     *BufferStr = (char *) arg;
92
93
0
  return tsCompareString(&BufferStr[a->pos], a->len,
94
0
               &BufferStr[b->pos], b->len,
95
0
               false);
96
0
}
97
98
/*
99
 * Sort an array of WordEntryIN, remove duplicates.
100
 * *outbuflen receives the amount of space needed for strings and positions.
101
 */
102
static int
103
uniqueentry(WordEntryIN *a, int l, char *buf, int *outbuflen)
104
0
{
105
0
  int     buflen;
106
0
  WordEntryIN *ptr,
107
0
         *res;
108
109
0
  Assert(l >= 1);
110
111
0
  if (l > 1)
112
0
    qsort_arg(a, l, sizeof(WordEntryIN), compareentry, buf);
113
114
0
  buflen = 0;
115
0
  res = a;
116
0
  ptr = a + 1;
117
0
  while (ptr - a < l)
118
0
  {
119
0
    if (!(ptr->entry.len == res->entry.len &&
120
0
        strncmp(&buf[ptr->entry.pos], &buf[res->entry.pos],
121
0
            res->entry.len) == 0))
122
0
    {
123
      /* done accumulating data into *res, count space needed */
124
0
      buflen += res->entry.len;
125
0
      if (res->entry.haspos)
126
0
      {
127
0
        res->poslen = uniquePos(res->pos, res->poslen);
128
0
        buflen = SHORTALIGN(buflen);
129
0
        buflen += res->poslen * sizeof(WordEntryPos) + sizeof(uint16);
130
0
      }
131
0
      res++;
132
0
      if (res != ptr)
133
0
        memcpy(res, ptr, sizeof(WordEntryIN));
134
0
    }
135
0
    else if (ptr->entry.haspos)
136
0
    {
137
0
      if (res->entry.haspos)
138
0
      {
139
        /* append ptr's positions to res's positions */
140
0
        int     newlen = ptr->poslen + res->poslen;
141
142
0
        res->pos = repalloc_array(res->pos, WordEntryPos, newlen);
143
0
        memcpy(&res->pos[res->poslen], ptr->pos,
144
0
             ptr->poslen * sizeof(WordEntryPos));
145
0
        res->poslen = newlen;
146
0
        pfree(ptr->pos);
147
0
      }
148
0
      else
149
0
      {
150
        /* just give ptr's positions to pos */
151
0
        res->entry.haspos = 1;
152
0
        res->pos = ptr->pos;
153
0
        res->poslen = ptr->poslen;
154
0
      }
155
0
    }
156
0
    ptr++;
157
0
  }
158
159
  /* count space needed for last item */
160
0
  buflen += res->entry.len;
161
0
  if (res->entry.haspos)
162
0
  {
163
0
    res->poslen = uniquePos(res->pos, res->poslen);
164
0
    buflen = SHORTALIGN(buflen);
165
0
    buflen += res->poslen * sizeof(WordEntryPos) + sizeof(uint16);
166
0
  }
167
168
0
  *outbuflen = buflen;
169
0
  return res + 1 - a;
170
0
}
171
172
173
Datum
174
tsvectorin(PG_FUNCTION_ARGS)
175
0
{
176
0
  char     *buf = PG_GETARG_CSTRING(0);
177
0
  Node     *escontext = fcinfo->context;
178
0
  TSVectorParseState state;
179
0
  WordEntryIN *arr;
180
0
  int     totallen;
181
0
  int     arrlen;     /* allocated size of arr */
182
0
  WordEntry  *inarr;
183
0
  int     len = 0;
184
0
  TSVector  in;
185
0
  int     i;
186
0
  char     *token;
187
0
  int     toklen;
188
0
  WordEntryPos *pos;
189
0
  int     poslen;
190
0
  char     *strbuf;
191
0
  int     stroff;
192
193
  /*
194
   * Tokens are appended to tmpbuf, cur is a pointer to the end of used
195
   * space in tmpbuf.
196
   */
197
0
  char     *tmpbuf;
198
0
  char     *cur;
199
0
  int     buflen = 256; /* allocated size of tmpbuf */
200
201
0
  state = init_tsvector_parser(buf, 0, escontext);
202
203
0
  arrlen = 64;
204
0
  arr = palloc_array(WordEntryIN, arrlen);
205
0
  cur = tmpbuf = palloc_array(char, buflen);
206
207
0
  while (gettoken_tsvector(state, &token, &toklen, &pos, &poslen, NULL))
208
0
  {
209
0
    if (toklen > MAXSTRLEN)
210
0
      ereturn(escontext, (Datum) 0,
211
0
          (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
212
0
           errmsg("word is too long (%d bytes, max %d bytes)",
213
0
              toklen,
214
0
              MAXSTRLEN)));
215
216
0
    if (cur - tmpbuf > MAXSTRPOS)
217
0
      ereturn(escontext, (Datum) 0,
218
0
          (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
219
0
           errmsg("string is too long for tsvector (%zu bytes, max %zu bytes)",
220
0
              (size_t) (cur - tmpbuf), (size_t) MAXSTRPOS)));
221
222
    /*
223
     * Enlarge buffers if needed
224
     */
225
0
    if (len >= arrlen)
226
0
    {
227
0
      arrlen *= 2;
228
0
      arr = repalloc_array(arr, WordEntryIN, arrlen);
229
0
    }
230
0
    while ((cur - tmpbuf) + toklen >= buflen)
231
0
    {
232
0
      int     dist = cur - tmpbuf;
233
234
0
      buflen *= 2;
235
0
      tmpbuf = (char *) repalloc(tmpbuf, buflen);
236
0
      cur = tmpbuf + dist;
237
0
    }
238
0
    arr[len].entry.len = toklen;
239
0
    arr[len].entry.pos = cur - tmpbuf;
240
0
    memcpy(cur, token, toklen);
241
0
    cur += toklen;
242
243
0
    if (poslen != 0)
244
0
    {
245
0
      arr[len].entry.haspos = 1;
246
0
      arr[len].pos = pos;
247
0
      arr[len].poslen = poslen;
248
0
    }
249
0
    else
250
0
    {
251
0
      arr[len].entry.haspos = 0;
252
0
      arr[len].pos = NULL;
253
0
      arr[len].poslen = 0;
254
0
    }
255
0
    len++;
256
0
  }
257
258
0
  close_tsvector_parser(state);
259
260
  /* Did gettoken_tsvector fail? */
261
0
  if (SOFT_ERROR_OCCURRED(escontext))
262
0
    PG_RETURN_NULL();
263
264
0
  if (len > 0)
265
0
    len = uniqueentry(arr, len, tmpbuf, &buflen);
266
0
  else
267
0
    buflen = 0;
268
269
0
  if (buflen > MAXSTRPOS)
270
0
    ereturn(escontext, (Datum) 0,
271
0
        (errcode(ERRCODE_PROGRAM_LIMIT_EXCEEDED),
272
0
         errmsg("string is too long for tsvector (%zu bytes, max %zu bytes)",
273
0
            (size_t) buflen, (size_t) MAXSTRPOS)));
274
275
0
  totallen = CALCDATASIZE(len, buflen);
276
0
  in = (TSVector) palloc0(totallen);
277
0
  SET_VARSIZE(in, totallen);
278
0
  in->size = len;
279
0
  inarr = ARRPTR(in);
280
0
  strbuf = STRPTR(in);
281
0
  stroff = 0;
282
0
  for (i = 0; i < len; i++)
283
0
  {
284
0
    memcpy(strbuf + stroff, &tmpbuf[arr[i].entry.pos], arr[i].entry.len);
285
0
    arr[i].entry.pos = stroff;
286
0
    stroff += arr[i].entry.len;
287
0
    if (arr[i].entry.haspos)
288
0
    {
289
      /* This should be unreachable because of MAXNUMPOS restrictions */
290
0
      if (arr[i].poslen > 0xFFFF)
291
0
        elog(ERROR, "positions array too long");
292
293
      /* Copy number of positions */
294
0
      stroff = SHORTALIGN(stroff);
295
0
      *(uint16 *) (strbuf + stroff) = (uint16) arr[i].poslen;
296
0
      stroff += sizeof(uint16);
297
298
      /* Copy positions */
299
0
      memcpy(strbuf + stroff, arr[i].pos, arr[i].poslen * sizeof(WordEntryPos));
300
0
      stroff += arr[i].poslen * sizeof(WordEntryPos);
301
302
0
      pfree(arr[i].pos);
303
0
    }
304
0
    inarr[i] = arr[i].entry;
305
0
  }
306
307
0
  Assert((strbuf + stroff - (char *) in) == totallen);
308
309
0
  PG_RETURN_TSVECTOR(in);
310
0
}
311
312
Datum
313
tsvectorout(PG_FUNCTION_ARGS)
314
0
{
315
0
  TSVector  out = PG_GETARG_TSVECTOR(0);
316
0
  char     *outbuf;
317
0
  int32   i,
318
0
        pp;
319
0
  size_t    lenbuf;
320
0
  WordEntry  *ptr = ARRPTR(out);
321
0
  char     *curin,
322
0
         *curout;
323
0
  const char *curend;
324
325
0
  lenbuf = out->size * 2 /* '' */ + out->size - 1 /* space */ + 2 /* \0 */ ;
326
0
  for (i = 0; i < out->size; i++)
327
0
  {
328
0
    lenbuf += ptr[i].len * 2 /* allow for escapes */ ;
329
0
    if (ptr[i].haspos)
330
0
      lenbuf += 1 /* : */ + 7 /* int2 + , + weight */ * POSDATALEN(out, &(ptr[i]));
331
0
  }
332
333
0
  curout = outbuf = (char *) palloc(lenbuf);
334
0
  for (i = 0; i < out->size; i++)
335
0
  {
336
0
    curin = STRPTR(out) + ptr->pos;
337
0
    curend = curin + ptr->len;
338
0
    if (i != 0)
339
0
      *curout++ = ' ';
340
0
    *curout++ = '\'';
341
0
    while (curin < curend)
342
0
    {
343
0
      int     len = pg_mblen_range(curin, curend);
344
345
0
      if (t_iseq(curin, '\''))
346
0
        *curout++ = '\'';
347
0
      else if (t_iseq(curin, '\\'))
348
0
        *curout++ = '\\';
349
350
0
      while (len--)
351
0
        *curout++ = *curin++;
352
0
    }
353
354
0
    *curout++ = '\'';
355
0
    if ((pp = POSDATALEN(out, ptr)) != 0)
356
0
    {
357
0
      WordEntryPos *wptr;
358
359
0
      *curout++ = ':';
360
0
      wptr = POSDATAPTR(out, ptr);
361
0
      while (pp)
362
0
      {
363
0
        curout += sprintf(curout, "%d", WEP_GETPOS(*wptr));
364
0
        switch (WEP_GETWEIGHT(*wptr))
365
0
        {
366
0
          case 3:
367
0
            *curout++ = 'A';
368
0
            break;
369
0
          case 2:
370
0
            *curout++ = 'B';
371
0
            break;
372
0
          case 1:
373
0
            *curout++ = 'C';
374
0
            break;
375
0
          case 0:
376
0
          default:
377
0
            break;
378
0
        }
379
380
0
        if (pp > 1)
381
0
          *curout++ = ',';
382
0
        pp--;
383
0
        wptr++;
384
0
      }
385
0
    }
386
0
    ptr++;
387
0
  }
388
389
0
  *curout = '\0';
390
0
  PG_FREE_IF_COPY(out, 0);
391
0
  PG_RETURN_CSTRING(outbuf);
392
0
}
393
394
/*
395
 * Binary Input / Output functions. The binary format is as follows:
396
 *
397
 * uint32 number of lexemes
398
 *
399
 * for each lexeme:
400
 *    lexeme text in client encoding, null-terminated
401
 *    uint16  number of positions
402
 *    for each position:
403
 *      uint16 WordEntryPos
404
 */
405
406
Datum
407
tsvectorsend(PG_FUNCTION_ARGS)
408
0
{
409
0
  TSVector  vec = PG_GETARG_TSVECTOR(0);
410
0
  StringInfoData buf;
411
0
  int     i,
412
0
        j;
413
0
  WordEntry  *weptr = ARRPTR(vec);
414
415
0
  pq_begintypsend(&buf);
416
417
0
  pq_sendint32(&buf, vec->size);
418
0
  for (i = 0; i < vec->size; i++)
419
0
  {
420
0
    uint16    npos;
421
422
    /*
423
     * the strings in the TSVector array are not null-terminated, so we
424
     * have to send the null-terminator separately
425
     */
426
0
    pq_sendtext(&buf, STRPTR(vec) + weptr->pos, weptr->len);
427
0
    pq_sendbyte(&buf, '\0');
428
429
0
    npos = POSDATALEN(vec, weptr);
430
0
    pq_sendint16(&buf, npos);
431
432
0
    if (npos > 0)
433
0
    {
434
0
      WordEntryPos *wepptr = POSDATAPTR(vec, weptr);
435
436
0
      for (j = 0; j < npos; j++)
437
0
        pq_sendint16(&buf, wepptr[j]);
438
0
    }
439
0
    weptr++;
440
0
  }
441
442
0
  PG_RETURN_BYTEA_P(pq_endtypsend(&buf));
443
0
}
444
445
Datum
446
tsvectorrecv(PG_FUNCTION_ARGS)
447
0
{
448
0
  StringInfo  buf = (StringInfo) PG_GETARG_POINTER(0);
449
0
  TSVector  vec;
450
0
  int     i;
451
0
  int32   nentries;
452
0
  int     datalen;    /* number of bytes used in the variable size
453
                 * area after fixed size TSVector header and
454
                 * WordEntries */
455
0
  Size    hdrlen;
456
0
  Size    len;      /* allocated size of vec */
457
0
  bool    needSort = false;
458
459
0
  nentries = pq_getmsgint(buf, sizeof(int32));
460
461
  /* We disallow empty lexemes, so more than MAXSTRPOS of them can't fit */
462
0
  if (nentries < 0 || nentries > MAXSTRPOS)
463
0
    elog(ERROR, "invalid size of tsvector");
464
465
0
  hdrlen = DATAHDRSIZE + sizeof(WordEntry) * nentries;
466
467
0
  len = hdrlen * 2;     /* times two to make some room for lexemes */
468
0
  vec = (TSVector) palloc0(len);
469
0
  vec->size = nentries;
470
471
0
  datalen = 0;
472
0
  for (i = 0; i < nentries; i++)
473
0
  {
474
0
    const char *lexeme;
475
0
    uint16    npos;
476
0
    size_t    lex_len;
477
478
0
    lexeme = pq_getmsgstring(buf);
479
0
    npos = (uint16) pq_getmsgint(buf, sizeof(uint16));
480
481
    /* sanity checks */
482
483
0
    lex_len = strlen(lexeme);
484
0
    if (lex_len == 0)
485
0
      elog(ERROR, "invalid tsvector: empty lexeme");
486
0
    if (lex_len > MAXSTRLEN)
487
0
      elog(ERROR, "invalid tsvector: lexeme too long");
488
489
0
    if (datalen > MAXSTRPOS)
490
0
      elog(ERROR, "invalid tsvector: maximum total lexeme length exceeded");
491
492
0
    if (npos > MAXNUMPOS)
493
0
      elog(ERROR, "unexpected number of tsvector positions");
494
495
    /*
496
     * Looks valid. Fill the WordEntry struct, and copy lexeme.
497
     *
498
     * But make sure the buffer is large enough first.
499
     */
500
0
    while (hdrlen + SHORTALIGN(datalen + lex_len) +
501
0
         sizeof(uint16) + npos * sizeof(WordEntryPos) >= len)
502
0
    {
503
0
      len *= 2;
504
0
      vec = (TSVector) repalloc(vec, len);
505
0
    }
506
507
0
    vec->entries[i].haspos = (npos > 0) ? 1 : 0;
508
0
    vec->entries[i].len = lex_len;
509
0
    vec->entries[i].pos = datalen;
510
511
0
    memcpy(STRPTR(vec) + datalen, lexeme, lex_len);
512
513
0
    datalen += lex_len;
514
515
0
    if (i > 0 && compareentry(&vec->entries[i],
516
0
                  &vec->entries[i - 1],
517
0
                  STRPTR(vec)) <= 0)
518
0
      needSort = true;
519
520
    /* Receive positions */
521
0
    if (npos > 0)
522
0
    {
523
0
      uint16    j;
524
0
      WordEntryPos *wepptr;
525
526
      /*
527
       * Pad to 2-byte alignment if necessary. Though we used palloc0
528
       * for the initial allocation, subsequent repalloc'd memory areas
529
       * are not initialized to zero.
530
       */
531
0
      if (datalen != SHORTALIGN(datalen))
532
0
      {
533
0
        *(STRPTR(vec) + datalen) = '\0';
534
0
        datalen = SHORTALIGN(datalen);
535
0
      }
536
537
0
      memcpy(STRPTR(vec) + datalen, &npos, sizeof(uint16));
538
539
0
      wepptr = POSDATAPTR(vec, &vec->entries[i]);
540
0
      for (j = 0; j < npos; j++)
541
0
      {
542
0
        wepptr[j] = (WordEntryPos) pq_getmsgint(buf, sizeof(WordEntryPos));
543
0
        if (j > 0 && WEP_GETPOS(wepptr[j]) <= WEP_GETPOS(wepptr[j - 1]))
544
0
          elog(ERROR, "position information is misordered");
545
0
      }
546
547
0
      datalen += sizeof(uint16) + npos * sizeof(WordEntryPos);
548
0
    }
549
0
  }
550
551
  /*
552
   * Enforce that datalen is still within MAXSTRPOS, ie the last lexeme
553
   * didn't go past that.  We could allow that, since no "pos" field
554
   * overflowed, but tsvectorrecv shouldn't accept values that other
555
   * tsvector-constructing routines wouldn't.
556
   */
557
0
  if (datalen > MAXSTRPOS)
558
0
    elog(ERROR, "invalid tsvector: maximum total lexeme length exceeded");
559
560
0
  SET_VARSIZE(vec, hdrlen + datalen);
561
562
0
  if (needSort)
563
0
    qsort_arg(ARRPTR(vec), vec->size, sizeof(WordEntry),
564
0
          compareentry, STRPTR(vec));
565
566
0
  PG_RETURN_TSVECTOR(vec);
567
0
}