Coverage Report

Created: 2026-09-14 07:34

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/ghostpdl/pdf/pdf_xref.c
Line
Count
Source
1
/* Copyright (C) 2018-2025 Artifex Software, Inc.
2
   All Rights Reserved.
3
4
   This software is provided AS-IS with no warranty, either express or
5
   implied.
6
7
   This software is distributed under license and may not be copied,
8
   modified or distributed except as expressly authorized under the terms
9
   of the license contained in the file LICENSE in this distribution.
10
11
   Refer to licensing information at http://www.artifex.com or contact
12
   Artifex Software, Inc.,  39 Mesa Street, Suite 108A, San Francisco,
13
   CA 94129, USA, for further information.
14
*/
15
16
/* xref parsing */
17
18
#include "pdf_int.h"
19
#include "pdf_stack.h"
20
#include "pdf_xref.h"
21
#include "pdf_file.h"
22
#include "pdf_loop_detect.h"
23
#include "pdf_dict.h"
24
#include "pdf_array.h"
25
#include "pdf_repair.h"
26
27
static int resize_xref(pdf_context *ctx, uint64_t new_size)
28
17.6k
{
29
17.6k
    xref_entry *new_xrefs;
30
31
    /* Although we can technically handle object numbers larger than this, on some systems (32-bit Windows)
32
     * memset is limited to a (signed!) integer for the size of memory to clear. We could deal
33
     * with this by clearing the memory in blocks, but really, this is almost certainly a
34
     * corrupted file or something.
35
     */
36
17.6k
    if (new_size >= (0x7ffffff / sizeof(xref_entry)))
37
22
        return_error(gs_error_rangecheck);
38
39
17.6k
    new_xrefs = (xref_entry *)gs_alloc_bytes(ctx->memory, (size_t)(new_size) * sizeof(xref_entry), "read_xref_stream allocate xref table entries");
40
17.6k
    if (new_xrefs == NULL){
41
0
        pdfi_countdown(ctx->xref_table);
42
0
        ctx->xref_table = NULL;
43
0
        return_error(gs_error_VMerror);
44
0
    }
45
17.6k
    memset(new_xrefs, 0x00, (new_size) * sizeof(xref_entry));
46
17.6k
    memcpy(new_xrefs, ctx->xref_table->xref, ctx->xref_table->xref_size * sizeof(xref_entry));
47
17.6k
    gs_free_object(ctx->memory, ctx->xref_table->xref, "reallocated xref entries");
48
17.6k
    ctx->xref_table->xref = new_xrefs;
49
17.6k
    ctx->xref_table->xref_size = new_size;
50
17.6k
    return 0;
51
17.6k
}
52
53
static int read_xref_stream_entries(pdf_context *ctx, pdf_c_stream *s, int64_t first, int64_t last, int64_t *W)
54
13.9k
{
55
13.9k
    uint i, j;
56
13.9k
    uint64_t field_width = 0;
57
13.9k
    uint32_t type = 0;
58
13.9k
    uint64_t objnum = 0, gen = 0;
59
13.9k
    byte *Buffer;
60
13.9k
    int64_t bytes = 0;
61
13.9k
    xref_entry *entry;
62
63
    /* Find max number of bytes to be read */
64
13.9k
    field_width = W[0];
65
13.9k
    if (W[1] > field_width)
66
13.8k
        field_width = W[1];
67
13.9k
    if (W[2] > field_width)
68
16
        field_width = W[2];
69
70
13.9k
    Buffer = gs_alloc_bytes(ctx->memory, field_width, "read_xref_stream_entry working buffer");
71
13.9k
    if (Buffer == NULL)
72
0
        return_error(gs_error_VMerror);
73
74
671k
    for (i=first;i<=last; i++){
75
        /* Defaults if W[n] = 0 */
76
657k
        type = 1;
77
657k
        objnum = gen = 0;
78
79
657k
        if (W[0] != 0) {
80
654k
            type = 0;
81
654k
            bytes = pdfi_read_bytes(ctx, Buffer, 1, W[0], s);
82
654k
            if (bytes < W[0]){
83
115
                gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry, free working buffer (error)");
84
115
                return_error(gs_error_ioerror);
85
115
            }
86
1.30M
            for (j=0;j<W[0];j++)
87
654k
                type = (type << 8) + Buffer[j];
88
654k
        }
89
90
657k
        if (W[1] != 0) {
91
657k
            bytes = pdfi_read_bytes(ctx, Buffer, 1, W[1], s);
92
657k
            if (bytes < W[1]){
93
27
                gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry free working buffer (error)");
94
27
                return_error(gs_error_ioerror);
95
27
            }
96
2.31M
            for (j=0;j<W[1];j++)
97
1.65M
                objnum = (objnum << 8) + Buffer[j];
98
657k
        }
99
100
657k
        if (W[2] != 0) {
101
643k
            bytes = pdfi_read_bytes(ctx, Buffer, 1, W[2], s);
102
643k
            if (bytes < W[2]){
103
32
                gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry, free working buffer (error)");
104
32
                return_error(gs_error_ioerror);
105
32
            }
106
1.30M
            for (j=0;j<W[2];j++)
107
660k
                gen = (gen << 8) + Buffer[j];
108
643k
        }
109
110
657k
        entry = &ctx->xref_table->xref[i];
111
657k
        if (entry->object_num != 0 && !entry->free)
112
3.77k
            continue;
113
114
653k
        entry->compressed = false;
115
653k
        entry->free = false;
116
653k
        entry->object_num = i;
117
653k
        entry->cache = NULL;
118
119
653k
        switch(type) {
120
17.6k
            case 0:
121
17.6k
                entry->free = true;
122
17.6k
                entry->u.uncompressed.offset = objnum;         /* For free objects we use the offset to store the object number of the next free object */
123
17.6k
                entry->u.uncompressed.generation_num = gen;    /* And the generation number is the numebr to use if this object is used again */
124
17.6k
                break;
125
219k
            case 1:
126
219k
                entry->u.uncompressed.offset = objnum;
127
219k
                entry->u.uncompressed.generation_num = gen;
128
219k
                break;
129
416k
            case 2:
130
416k
                entry->compressed = true;
131
416k
                entry->u.compressed.compressed_stream_num = objnum;   /* The object number of the compressed stream */
132
416k
                entry->u.compressed.object_index = gen;               /* And the index of the object within the stream */
133
416k
                break;
134
122
            default:
135
122
                gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry, free working buffer");
136
122
                return_error(gs_error_rangecheck);
137
0
                break;
138
653k
        }
139
653k
    }
140
13.6k
    gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry, free working buffer");
141
13.6k
    return 0;
142
13.9k
}
143
144
/* Forward definition */
145
static int read_xref(pdf_context *ctx, pdf_c_stream *s);
146
static int pdfi_check_xref_stream(pdf_context *ctx);
147
/* These two routines are recursive.... */
148
static int pdfi_read_xref_stream_dict(pdf_context *ctx, pdf_c_stream *s, int obj_num);
149
150
static int pdfi_process_xref_stream(pdf_context *ctx, pdf_stream *stream_obj, pdf_c_stream *s)
151
11.1k
{
152
11.1k
    pdf_c_stream *XRefStrm;
153
11.1k
    int code, i;
154
11.1k
    pdf_dict *sdict = NULL;
155
11.1k
    pdf_name *n;
156
11.1k
    pdf_array *a;
157
11.1k
    int64_t size;
158
11.1k
    int64_t num;
159
11.1k
    int64_t W[3] = {0, 0, 0};
160
11.1k
    int objnum;
161
11.1k
    bool known = false;
162
163
11.1k
    if (pdfi_type_of(stream_obj) != PDF_STREAM)
164
0
        return_error(gs_error_typecheck);
165
166
11.1k
    code = pdfi_dict_from_obj(ctx, (pdf_obj *)stream_obj, &sdict);
167
11.1k
    if (code < 0)
168
0
        return code;
169
170
11.1k
    code = pdfi_dict_get_type(ctx, sdict, "Type", PDF_NAME, (pdf_obj **)&n);
171
11.1k
    if (code < 0)
172
45
        return code;
173
174
11.0k
    if (n->length != 4 || memcmp(n->data, "XRef", 4) != 0) {
175
22
        pdfi_countdown(n);
176
22
        return_error(gs_error_syntaxerror);
177
22
    }
178
11.0k
    pdfi_countdown(n);
179
180
11.0k
    code = pdfi_dict_get_int(ctx, sdict, "Size", &size);
181
11.0k
    if (code < 0)
182
10
        return code;
183
11.0k
    if (size < 1)
184
9
        return 0;
185
186
11.0k
    if (size < 0 || size > floor((double)ARCH_MAX_SIZE_T / (double)sizeof(xref_entry)))
187
0
        return_error(gs_error_rangecheck);
188
189
    /* If this is the first xref stream then allocate the xref table and store the trailer */
190
11.0k
    if (ctx->xref_table == NULL) {
191
7.03k
        ctx->xref_table = (xref_table_t *)gs_alloc_bytes(ctx->memory, sizeof(xref_table_t), "read_xref_stream allocate xref table");
192
7.03k
        if (ctx->xref_table == NULL) {
193
0
            return_error(gs_error_VMerror);
194
0
        }
195
7.03k
        memset(ctx->xref_table, 0x00, sizeof(xref_table_t));
196
7.03k
        ctx->xref_table->xref = (xref_entry *)gs_alloc_bytes(ctx->memory, (size_t)size * sizeof(xref_entry), "read_xref_stream allocate xref table entries");
197
7.03k
        if (ctx->xref_table->xref == NULL){
198
2
            gs_free_object(ctx->memory, ctx->xref_table, "failed to allocate xref table entries");
199
2
            ctx->xref_table = NULL;
200
2
            return_error(gs_error_VMerror);
201
2
        }
202
7.03k
        memset(ctx->xref_table->xref, 0x00, size * sizeof(xref_entry));
203
7.03k
        ctx->xref_table->ctx = ctx;
204
7.03k
        ctx->xref_table->type = PDF_XREF_TABLE;
205
7.03k
        ctx->xref_table->xref_size = size;
206
#if REFCNT_DEBUG
207
        ctx->xref_table->UID = ctx->ref_UID++;
208
        outprintf(ctx->memory, "Allocated xref table with UID %"PRIi64"\n", ctx->xref_table->UID);
209
#endif
210
7.03k
        pdfi_countup(ctx->xref_table);
211
212
7.03k
        pdfi_countdown(ctx->Trailer);
213
214
7.03k
        ctx->Trailer = sdict;
215
7.03k
        pdfi_countup(sdict);
216
7.03k
    } else {
217
4.00k
        if (size > ctx->xref_table->xref_size)
218
3
            return_error(gs_error_rangecheck);
219
220
3.99k
        code = pdfi_merge_dicts(ctx, ctx->Trailer, sdict);
221
3.99k
        if (code < 0 && (code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREF, "pdfi_process_xref_stream", NULL)) < 0) {
222
0
            goto exit;
223
0
        }
224
3.99k
    }
225
226
11.0k
    pdfi_seek(ctx, ctx->main_stream, pdfi_stream_offset(ctx, stream_obj), SEEK_SET);
227
228
    /* Bug #691220 has a PDF file with a compressed XRef, the stream dictionary has
229
     * a /DecodeParms entry for the stream, which has a /Colors value of 5, which makes
230
     * *no* sense whatever. If we try to apply a Predictor then we end up in a loop trying
231
     * to read 5 colour samples. Rather than meddles with more parameters to the filter
232
     * code, we'll just remove the Colors entry from the DecodeParms dictionary,
233
     * because it is nonsense. This means we'll get the (sensible) default value of 1.
234
     */
235
11.0k
    code = pdfi_dict_known(ctx, sdict, "DecodeParms", &known);
236
11.0k
    if (code < 0)
237
0
        return code;
238
239
11.0k
    if (known) {
240
10.1k
        pdf_dict *DP;
241
10.1k
        double f;
242
10.1k
        pdf_obj *name;
243
244
10.1k
        code = pdfi_dict_get_type(ctx, sdict, "DecodeParms", PDF_DICT, (pdf_obj **)&DP);
245
10.1k
        if (code < 0)
246
0
            return code;
247
248
10.1k
        code = pdfi_dict_knownget_number(ctx, DP, "Colors", &f);
249
10.1k
        if (code < 0) {
250
0
            pdfi_countdown(DP);
251
0
            return code;
252
0
        }
253
10.1k
        if (code > 0 && f != (double)1)
254
0
        {
255
0
            code = pdfi_name_alloc(ctx, (byte *)"Colors", 6, &name);
256
0
            if (code < 0) {
257
0
                pdfi_countdown(DP);
258
0
                return code;
259
0
            }
260
0
            pdfi_countup(name);
261
262
0
            code = pdfi_dict_delete_pair(ctx, DP, (pdf_name *)name);
263
0
            pdfi_countdown(name);
264
0
            if (code < 0) {
265
0
                pdfi_countdown(DP);
266
0
                return code;
267
0
            }
268
0
        }
269
10.1k
        pdfi_countdown(DP);
270
10.1k
    }
271
272
11.0k
    code = pdfi_filter_no_decryption(ctx, stream_obj, s, &XRefStrm, false);
273
11.0k
    if (code < 0) {
274
43
        pdfi_countdown(ctx->xref_table);
275
43
        ctx->xref_table = NULL;
276
43
        return code;
277
43
    }
278
279
10.9k
    code = pdfi_dict_get_type(ctx, sdict, "W", PDF_ARRAY, (pdf_obj **)&a);
280
10.9k
    if (code < 0) {
281
7
        pdfi_close_file(ctx, XRefStrm);
282
7
        pdfi_countdown(ctx->xref_table);
283
7
        ctx->xref_table = NULL;
284
7
        return code;
285
7
    }
286
287
10.9k
    if (pdfi_array_size(a) != 3) {
288
10
        pdfi_countdown(a);
289
10
        pdfi_close_file(ctx, XRefStrm);
290
10
        pdfi_countdown(ctx->xref_table);
291
10
        ctx->xref_table = NULL;
292
10
        return_error(gs_error_rangecheck);
293
10
    }
294
43.8k
    for (i=0;i<3;i++) {
295
32.8k
        code = pdfi_array_get_int(ctx, a, (uint64_t)i, (int64_t *)&W[i]);
296
32.8k
        if (code < 0 || W[i] < 0) {
297
32
            pdfi_countdown(a);
298
32
            pdfi_close_file(ctx, XRefStrm);
299
32
            pdfi_countdown(ctx->xref_table);
300
32
            ctx->xref_table = NULL;
301
32
            if (W[i] < 0)
302
10
                code = gs_note_error(gs_error_rangecheck);
303
32
            return code;
304
32
        }
305
32.8k
    }
306
10.9k
    pdfi_countdown(a);
307
308
    /* W[0] is either:
309
     * 0 (no type field) or a single byte with the type.
310
     * W[1] is either:
311
     * The object number of the next free object, the byte offset of this object in the file or the object5 number of the object stream where this object is stored.
312
     * W[2] is either:
313
     * The generation number to use if this object is used again, the generation number of the object or the index of this object within the object stream.
314
     *
315
     * Object and generation numbers are limited to unsigned 64-bit values, as are bytes offsets in the file, indexes of objects within the stream likewise (actually
316
     * most of these are generally 32-bit max). So we can limit the field widths to 8 bytes, enough to hold a 64-bit number.
317
     * Even if a later version of the spec makes these larger (which seems unlikely!) we still cna't cope with integers > 64-bits.
318
     */
319
10.9k
    if (W[0] > 1 || W[1] > 8 || W[2] > 8) {
320
28
        pdfi_close_file(ctx, XRefStrm);
321
28
        pdfi_countdown(ctx->xref_table);
322
28
        ctx->xref_table = NULL;
323
28
        return code;
324
28
    }
325
326
10.9k
    code = pdfi_dict_get_type(ctx, sdict, "Index", PDF_ARRAY, (pdf_obj **)&a);
327
10.9k
    if (code == gs_error_undefined) {
328
3.89k
        code = read_xref_stream_entries(ctx, XRefStrm, 0, size - 1, W);
329
3.89k
        if (code < 0) {
330
93
            pdfi_close_file(ctx, XRefStrm);
331
93
            pdfi_countdown(ctx->xref_table);
332
93
            ctx->xref_table = NULL;
333
93
            return code;
334
93
        }
335
7.01k
    } else {
336
7.01k
        int64_t start, size;
337
338
7.01k
        if (code < 0) {
339
3
            pdfi_close_file(ctx, XRefStrm);
340
3
            pdfi_countdown(ctx->xref_table);
341
3
            ctx->xref_table = NULL;
342
3
            return code;
343
3
        }
344
345
7.00k
        if (pdfi_array_size(a) & 1) {
346
11
            pdfi_countdown(a);
347
11
            pdfi_close_file(ctx, XRefStrm);
348
11
            pdfi_countdown(ctx->xref_table);
349
11
            ctx->xref_table = NULL;
350
11
            return_error(gs_error_rangecheck);
351
11
        }
352
353
16.8k
        for (i=0;i < pdfi_array_size(a);i+=2){
354
10.0k
            code = pdfi_array_get_int(ctx, a, (uint64_t)i, &start);
355
10.0k
            if (code < 0 || start < 0) {
356
14
                pdfi_countdown(a);
357
14
                pdfi_close_file(ctx, XRefStrm);
358
14
                pdfi_countdown(ctx->xref_table);
359
14
                ctx->xref_table = NULL;
360
14
                return code;
361
14
            }
362
363
10.0k
            code = pdfi_array_get_int(ctx, a, (uint64_t)i+1, &size);
364
10.0k
            if (code < 0) {
365
12
                pdfi_countdown(a);
366
12
                pdfi_close_file(ctx, XRefStrm);
367
12
                pdfi_countdown(ctx->xref_table);
368
12
                ctx->xref_table = NULL;
369
12
                return code;
370
12
            }
371
372
10.0k
            if (size < 1)
373
11
                continue;
374
375
10.0k
            if (start + size >= ctx->xref_table->xref_size) {
376
6.38k
                code = resize_xref(ctx, start + size);
377
6.38k
                if (code < 0) {
378
6
                    pdfi_countdown(a);
379
6
                    pdfi_close_file(ctx, XRefStrm);
380
6
                    pdfi_countdown(ctx->xref_table);
381
6
                    ctx->xref_table = NULL;
382
6
                    return code;
383
6
                }
384
6.38k
            }
385
386
10.0k
            code = read_xref_stream_entries(ctx, XRefStrm, start, start + size - 1, W);
387
10.0k
            if (code < 0) {
388
203
                pdfi_countdown(a);
389
203
                pdfi_close_file(ctx, XRefStrm);
390
203
                pdfi_countdown(ctx->xref_table);
391
203
                ctx->xref_table = NULL;
392
203
                return code;
393
203
            }
394
10.0k
        }
395
6.99k
    }
396
10.5k
    pdfi_countdown(a);
397
398
10.5k
    pdfi_close_file(ctx, XRefStrm);
399
400
10.5k
    code = pdfi_dict_get_int(ctx, sdict, "Prev", &num);
401
10.5k
    if (code == gs_error_undefined)
402
4.30k
        return 0;
403
404
6.26k
    if (code < 0)
405
11
        return code;
406
407
6.25k
    if (num < 0 || num > ctx->main_stream_length)
408
1.90k
        return_error(gs_error_rangecheck);
409
410
4.34k
    if (pdfi_loop_detector_check_object(ctx, num) == true)
411
7
        return_error(gs_error_circular_reference);
412
4.33k
    else {
413
4.33k
        code = pdfi_loop_detector_add_object(ctx, num);
414
4.33k
        if (code < 0)
415
0
            return code;
416
4.33k
    }
417
418
4.33k
    if(ctx->args.pdfdebug)
419
0
        outprintf(ctx->memory, "%% Reading /Prev xref\n");
420
421
4.33k
    pdfi_seek(ctx, s, num, SEEK_SET);
422
423
4.33k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &objnum);
424
4.33k
    if (code == 1) {
425
4.04k
        if (pdfi_check_xref_stream(ctx))
426
4.01k
            return pdfi_read_xref_stream_dict(ctx, s, objnum);
427
4.04k
    }
428
429
323
    code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
430
323
    if (code < 0)
431
0
        return code;
432
323
    if (code == TOKEN_XREF) {
433
24
        if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_syntaxerror), NULL, E_PDF_PREV_NOT_XREF_STREAM, "pdfi_process_xref_stream", NULL)) < 0) {
434
0
            goto exit;
435
0
        }
436
        /* Read old-style xref table */
437
24
        return(read_xref(ctx, ctx->main_stream));
438
24
    }
439
299
exit:
440
299
    return_error(gs_error_syntaxerror);
441
323
}
442
443
static int pdfi_check_xref_stream(pdf_context *ctx)
444
14.0k
{
445
14.0k
    gs_offset_t offset;
446
14.0k
    int gen_num, code = 0;
447
448
14.0k
    offset = pdfi_unread_tell(ctx);
449
450
14.0k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &gen_num);
451
14.0k
    if (code <= 0) {
452
1.02k
        code = 0;
453
1.02k
        goto exit;
454
1.02k
    }
455
456
    /* Try to read 'obj' */
457
13.0k
    code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
458
13.0k
    if (code <= 0) {
459
0
        code = 0;
460
0
        goto exit;
461
0
    }
462
463
    /* Third element must be obj, or it's not a valid xref */
464
13.0k
    if (code != TOKEN_OBJ)
465
1.52k
        code = 0;
466
11.5k
    else
467
11.5k
        code = 1;
468
469
14.0k
exit:
470
14.0k
    pdfi_seek(ctx, ctx->main_stream, offset, SEEK_SET);
471
14.0k
    return code;
472
13.0k
}
473
474
static int pdfi_read_xref_stream_dict(pdf_context *ctx, pdf_c_stream *s, int obj_num)
475
11.6k
{
476
11.6k
    int code;
477
11.6k
    int gen_num;
478
479
11.6k
    if (ctx->args.pdfdebug)
480
0
        outprintf(ctx->memory, "\n%% Reading PDF 1.5+ xref stream\n");
481
482
    /* We have the obj_num. Lets try for obj_num gen obj as a XRef stream */
483
11.6k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &gen_num);
484
11.6k
    if (code <= 0) {
485
0
        if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREFSTREAM, "pdfi_read_xref_stream_dict", "")) < 0) {
486
0
            return code;
487
0
        }
488
0
        return(pdfi_repair_file(ctx));
489
0
    }
490
491
    /* Try to read 'obj' */
492
11.6k
    code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
493
11.6k
    if (code < 0)
494
0
        return code;
495
11.6k
    if (code == 0)
496
0
        return_error(gs_error_syntaxerror);
497
498
    /* Third element must be obj, or it's not a valid xref */
499
11.6k
    if (code != TOKEN_OBJ) {
500
0
        if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_rangecheck), NULL, E_PDF_BAD_XREFSTMOFFSET, "pdfi_read_xref_stream_dict", "")) < 0) {
501
0
            return code;
502
0
        }
503
0
        return(pdfi_repair_file(ctx));
504
0
    }
505
506
479k
    do {
507
479k
        code = pdfi_read_token(ctx, ctx->main_stream, obj_num, gen_num);
508
479k
        if (code <= 0) {
509
410
            if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREFSTREAM, "pdfi_read_xref_stream_dict", NULL)) < 0) {
510
0
                return code;
511
0
            }
512
410
            return pdfi_repair_file(ctx);
513
410
        }
514
515
479k
        if (pdfi_count_stack(ctx) >= 2 && pdfi_type_of(ctx->stack_top[-1]) == PDF_FAST_KEYWORD) {
516
12.9k
            uintptr_t keyword = (uintptr_t)ctx->stack_top[-1];
517
12.9k
            if (keyword == TOKEN_STREAM) {
518
11.1k
                pdf_dict *dict;
519
11.1k
                pdf_stream *sdict = NULL;
520
11.1k
                int64_t Length;
521
522
                /* Remove the 'stream' token from the stack, should leave a dictionary object on the stack */
523
11.1k
                pdfi_pop(ctx, 1);
524
11.1k
                if (pdfi_type_of(ctx->stack_top[-1]) != PDF_DICT) {
525
21
                    if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREFSTREAM, "pdfi_read_xref_stream_dict", NULL)) < 0) {
526
0
                        return code;
527
0
                    }
528
21
                    return pdfi_repair_file(ctx);
529
21
                }
530
11.1k
                dict = (pdf_dict *)ctx->stack_top[-1];
531
532
                /* Convert the dict into a stream (sdict comes back with at least one ref) */
533
11.1k
                code = pdfi_obj_dict_to_stream(ctx, dict, &sdict, true);
534
                /* Pop off the dict */
535
11.1k
                pdfi_pop(ctx, 1);
536
11.1k
                if (code < 0) {
537
0
                    if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREFSTREAM, "pdfi_read_xref_stream_dict", NULL)) < 0) {
538
0
                        return code;
539
0
                    }
540
                    /* TODO: should I return code instead of trying to repair?
541
                     * Normally the above routine should not fail so something is
542
                     * probably seriously fubar.
543
                     */
544
0
                    return pdfi_repair_file(ctx);
545
0
                }
546
11.1k
                dict = NULL;
547
548
                /* Init the stuff for the stream */
549
11.1k
                sdict->stream_offset = pdfi_unread_tell(ctx);
550
11.1k
                sdict->object_num = obj_num;
551
11.1k
                sdict->generation_num = gen_num;
552
553
11.1k
                code = pdfi_dict_get_int(ctx, sdict->stream_dict, "Length", &Length);
554
11.1k
                if (code < 0) {
555
                    /* TODO: Not positive this will actually have a length -- just use 0 */
556
40
                    (void)pdfi_set_error_var(ctx, 0, NULL, E_PDF_BADSTREAM, "pdfi_read_xref_stream_dict", "Xref Stream object %u missing mandatory keyword /Length\n", obj_num);
557
40
                    code = 0;
558
40
                    Length = 0;
559
40
                }
560
11.1k
                sdict->Length = Length;
561
11.1k
                sdict->length_valid = true;
562
563
11.1k
                code = pdfi_process_xref_stream(ctx, sdict, ctx->main_stream);
564
11.1k
                pdfi_countdown(sdict);
565
11.1k
                if (code < 0) {
566
2.90k
                    pdfi_set_error(ctx, gs_note_error(gs_error_syntaxerror), NULL, E_PDF_PREV_NOT_XREF_STREAM, "pdfi_read_xref_stream_dict", NULL);
567
2.90k
                    return code;
568
2.90k
                }
569
8.21k
                break;
570
11.1k
            } else if (keyword == TOKEN_ENDOBJ) {
571
                /* Something went wrong, this is not a stream dictionary */
572
125
                if ((code = pdfi_set_error_var(ctx, 0, NULL, E_PDF_BADSTREAM, "pdfi_read_xref_stream_dict", "Xref Stream object %u missing mandatory keyword /Length\n", obj_num)) < 0) {
573
0
                    return code;
574
0
                }
575
125
                return(pdfi_repair_file(ctx));
576
125
            }
577
12.9k
        }
578
479k
    } while(1);
579
8.21k
    return 0;
580
11.6k
}
581
582
static int skip_to_digit(pdf_context *ctx, pdf_c_stream *s, unsigned int limit)
583
2.60k
{
584
2.60k
    int c, read = 0;
585
586
9.72k
    do {
587
9.72k
        c = pdfi_read_byte(ctx, s);
588
9.72k
        if (c < 0)
589
0
            return_error(gs_error_ioerror);
590
9.72k
        if (c >= '0' && c <= '9') {
591
2.34k
            pdfi_unread_byte(ctx, s, (byte)c);
592
2.34k
            return read;
593
2.34k
        }
594
7.37k
        read++;
595
7.37k
    } while (read < limit);
596
597
255
    return read;
598
2.60k
}
599
600
static int read_digits(pdf_context *ctx, pdf_c_stream *s, byte *Buffer, int limit)
601
2.60k
{
602
2.60k
    int c, read = 0;
603
604
    /* Since the "limit" is a value calculated by the caller,
605
       it's easier to check it in one place (here) than before
606
       every call.
607
     */
608
2.60k
    if (limit <= 0)
609
263
        return_error(gs_error_syntaxerror);
610
611
    /* We assume that Buffer always has limit+1 bytes available, so we can
612
     * safely terminate it. */
613
614
14.0k
    do {
615
14.0k
        c = pdfi_read_byte(ctx, s);
616
14.0k
        if (c < 0)
617
0
            return_error(gs_error_ioerror);
618
14.0k
        if (c < '0' || c > '9') {
619
1.01k
            pdfi_unread_byte(ctx, s, c);
620
1.01k
            break;
621
1.01k
        }
622
13.0k
        *Buffer++ = (byte)c;
623
13.0k
        read++;
624
13.0k
    } while (read < limit);
625
2.34k
    *Buffer = 0;
626
627
2.34k
    return read;
628
2.34k
}
629
630
631
static int read_xref_entry_slow(pdf_context *ctx, pdf_c_stream *s, gs_offset_t *offset, uint32_t *generation_num, unsigned char *free)
632
1.32k
{
633
1.32k
    byte Buffer[20];
634
1.32k
    int c, code, read = 0;
635
636
    /* First off, find a number. If we don't find one, and read 20 bytes, throw an error */
637
1.32k
    code = skip_to_digit(ctx, s, 20);
638
1.32k
    if (code < 0)
639
0
        return code;
640
1.32k
    read += code;
641
642
    /* Now read a number */
643
1.32k
    code = read_digits(ctx, s, (byte *)&Buffer,  (read > 10 ? 20 - read : 10));
644
1.32k
    if (code < 0)
645
41
        return code;
646
1.28k
    read += code;
647
648
1.28k
    *offset = atol((const char *)Buffer);
649
650
    /* find next number */
651
1.28k
    code = skip_to_digit(ctx, s, 20 - read);
652
1.28k
    if (code < 0)
653
0
        return code;
654
1.28k
    read += code;
655
656
    /* and read it */
657
1.28k
    code = read_digits(ctx, s, (byte *)&Buffer, (read > 15 ? 20 - read : 5));
658
1.28k
    if (code < 0)
659
222
        return code;
660
1.05k
    read += code;
661
662
1.05k
    *generation_num = atol((const char *)Buffer);
663
664
1.82k
    do {
665
1.82k
        c = pdfi_read_byte(ctx, s);
666
1.82k
        if (c < 0)
667
0
            return_error(gs_error_ioerror);
668
1.82k
        read ++;
669
1.82k
        if (c == 0x09 || c == 0x20)
670
799
            continue;
671
1.02k
        if (c == 'n' || c == 'f') {
672
577
            *free = (unsigned char)c;
673
577
            break;
674
577
        } else {
675
450
            return_error(gs_error_syntaxerror);
676
450
        }
677
1.02k
    } while (read < 20);
678
609
    if (read >= 20)
679
39
        return_error(gs_error_syntaxerror);
680
681
1.44k
    do {
682
1.44k
        c = pdfi_read_byte(ctx, s);
683
1.44k
        if (c < 0)
684
0
            return_error(gs_error_syntaxerror);
685
1.44k
        read++;
686
1.44k
        if (c == 0x20 || c == 0x09 || c == 0x0d || c == 0x0a)
687
727
            continue;
688
1.44k
    } while (read < 20);
689
570
    return 0;
690
570
}
691
692
static int write_offset(byte *B, gs_offset_t o, unsigned int g, unsigned char free)
693
570
{
694
570
    byte b[20], *ptr = B;
695
570
    int index = 0;
696
697
570
    gs_snprintf((char *)b, sizeof(b), "%"PRIdOFFSET"", o);
698
570
    if (strlen((const char *)b) > 10)
699
0
        return_error(gs_error_rangecheck);
700
4.66k
    for(index=0;index < 10 - strlen((const char *)b); index++) {
701
4.09k
        *ptr++ = 0x30;
702
4.09k
    }
703
570
    memcpy(ptr, b, strlen((const char *)b));
704
570
    ptr += strlen((const char *)b);
705
570
    *ptr++ = 0x20;
706
707
570
    gs_snprintf((char *)b, sizeof(b), "%d", g);
708
570
    if (strlen((const char *)b) > 5)
709
0
        return_error(gs_error_rangecheck);
710
2.59k
    for(index=0;index < 5 - strlen((const char *)b);index++) {
711
2.02k
        *ptr++ = 0x30;
712
2.02k
    }
713
570
    memcpy(ptr, b, strlen((const char *)b));
714
570
    ptr += strlen((const char *)b);
715
570
    *ptr++ = 0x20;
716
570
    *ptr++ = free;
717
570
    *ptr++ = 0x20;
718
570
    *ptr++ = 0x0d;
719
570
    return 0;
720
570
}
721
722
static int read_xref_section(pdf_context *ctx, pdf_c_stream *s, uint64_t *section_start, uint64_t *section_size)
723
31.8k
{
724
31.8k
    int code = 0, i, j;
725
31.8k
    int start = 0;
726
31.8k
    int size = 0;
727
31.8k
    int64_t bytes = 0;
728
31.8k
    char Buffer[21];
729
730
31.8k
    *section_start = *section_size = 0;
731
732
31.8k
    if (ctx->args.pdfdebug)
733
0
        outprintf(ctx->memory, "\n%% Reading xref section\n");
734
735
31.8k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &start);
736
31.8k
    if (code < 0) {
737
        /* Not an int, might be a keyword */
738
8.57k
        code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
739
8.57k
        if (code < 0)
740
0
            return code;
741
742
8.57k
        if (code != TOKEN_TRAILER) {
743
            /* element is not an integer, and not a keyword - not a valid xref */
744
118
            return_error(gs_error_typecheck);
745
118
        }
746
8.45k
        return 1;
747
8.57k
    }
748
749
23.2k
    if (start < 0)
750
13
        return_error(gs_error_rangecheck);
751
752
23.2k
    *section_start = start;
753
754
23.2k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &size);
755
23.2k
    if (code < 0)
756
17
        return code;
757
23.2k
    if (code == 0)
758
38
        return_error(gs_error_syntaxerror);
759
760
    /* Zero sized xref sections are valid; see the file attached to
761
     * bug 704947 for an example. */
762
23.2k
    if (size < 0)
763
10
        return_error(gs_error_rangecheck);
764
765
23.1k
    *section_size = size;
766
767
23.1k
    if (ctx->args.pdfdebug)
768
0
        outprintf(ctx->memory, "\n%% Section starts at %d and has %d entries\n", (unsigned int) start, (unsigned int)size);
769
770
23.1k
    if (size > 0) {
771
22.8k
        if (ctx->xref_table == NULL) {
772
8.40k
            ctx->xref_table = (xref_table_t *)gs_alloc_bytes(ctx->memory, sizeof(xref_table_t), "read_xref_stream allocate xref table");
773
8.40k
            if (ctx->xref_table == NULL)
774
0
                return_error(gs_error_VMerror);
775
8.40k
            memset(ctx->xref_table, 0x00, sizeof(xref_table_t));
776
777
8.40k
            ctx->xref_table->xref = (xref_entry *)gs_alloc_bytes(ctx->memory, ((size_t)start + (size_t)size) * (size_t)sizeof(xref_entry), "read_xref_stream allocate xref table entries");
778
8.40k
            if (ctx->xref_table->xref == NULL){
779
23
                gs_free_object(ctx->memory, ctx->xref_table, "free xref table on error allocating entries");
780
23
                ctx->xref_table = NULL;
781
23
                return_error(gs_error_VMerror);
782
23
            }
783
#if REFCNT_DEBUG
784
            ctx->xref_table->UID = ctx->ref_UID++;
785
            outprintf(ctx->memory, "Allocated xref table with UID %"PRIi64"\n", ctx->xref_table->UID);
786
#endif
787
788
8.37k
            memset(ctx->xref_table->xref, 0x00, (start + size) * sizeof(xref_entry));
789
8.37k
            ctx->xref_table->ctx = ctx;
790
8.37k
            ctx->xref_table->type = PDF_XREF_TABLE;
791
8.37k
            ctx->xref_table->xref_size = start + size;
792
8.37k
            pdfi_countup(ctx->xref_table);
793
14.4k
        } else {
794
14.4k
            if (start + size > ctx->xref_table->xref_size) {
795
11.2k
                code = resize_xref(ctx, start + size);
796
11.2k
                if (code < 0)
797
16
                    return code;
798
11.2k
            }
799
14.4k
        }
800
22.8k
    }
801
802
23.1k
    pdfi_skip_white(ctx, s);
803
309k
    for (i=0;i< size;i++){
804
287k
        xref_entry *entry = &ctx->xref_table->xref[i + start];
805
287k
        unsigned char free;
806
287k
        gs_offset_t off;
807
287k
        unsigned int gen;
808
809
287k
        bytes = pdfi_read_bytes(ctx, (byte *)Buffer, 1, 20, s);
810
287k
        if (bytes < 20)
811
2
            return_error(gs_error_ioerror);
812
287k
        j = 19;
813
287k
        if ((Buffer[19] != 0x0a && Buffer[19] != 0x0d) || (Buffer[18] != 0x0d && Buffer[18] != 0x0a && Buffer[18] != 0x20))
814
13.0k
            pdfi_set_warning(ctx, 0, NULL, W_PDF_BAD_XREF_ENTRY_SIZE, "read_xref_section", NULL);
815
305k
        while (Buffer[j] != 0x0D && Buffer[j] != 0x0A) {
816
19.2k
            pdfi_unread_byte(ctx, s, (byte)Buffer[j]);
817
19.2k
            if (--j < 0) {
818
630
                pdfi_set_warning(ctx, 0, NULL, W_PDF_BAD_XREF_ENTRY_NO_EOL, "read_xref_section", NULL);
819
630
                outprintf(ctx->memory, "Invalid xref entry, line terminator missing.\n");
820
630
                code = read_xref_entry_slow(ctx, s, &off, &gen, &free);
821
630
                if (code < 0)
822
337
                    return code;
823
293
                code = write_offset((byte *)Buffer, off, gen, free);
824
293
                if (code < 0)
825
0
                    return code;
826
293
                j = 19;
827
293
                break;
828
293
            }
829
19.2k
        }
830
286k
        Buffer[j] = 0x00;
831
286k
        if (entry->object_num != 0)
832
3.77k
            continue;
833
834
283k
        if (sscanf(Buffer, "%"PRIdOFFSET" %d %c", &entry->u.uncompressed.offset, &entry->u.uncompressed.generation_num, &free) != 3) {
835
692
            pdfi_set_warning(ctx, 0, NULL, W_PDF_BAD_XREF_ENTRY_FORMAT, "read_xref_section", NULL);
836
692
            outprintf(ctx->memory, "Invalid xref entry, incorrect format.\n");
837
692
            pdfi_unread(ctx, s, (byte *)Buffer, 20);
838
692
            code = read_xref_entry_slow(ctx, s, &off, &gen, &free);
839
692
            if (code < 0)
840
415
                return code;
841
277
            code = write_offset((byte *)Buffer, off, gen, free);
842
277
            if (code < 0)
843
0
                return code;
844
277
        }
845
846
282k
        entry->compressed = false;
847
282k
        entry->object_num = i + start;
848
282k
        if (free == 'f')
849
70.8k
            entry->free = true;
850
282k
        if(free == 'n')
851
211k
            entry->free = false;
852
282k
        if (entry->object_num == 0) {
853
5.79k
            if (!entry->free) {
854
60
                pdfi_set_warning(ctx, 0, NULL, W_PDF_XREF_OBJECT0_NOT_FREE, "read_xref_section", NULL);
855
60
            }
856
5.79k
        }
857
282k
    }
858
859
22.4k
    return 0;
860
23.1k
}
861
862
static int read_xref(pdf_context *ctx, pdf_c_stream *s)
863
9.44k
{
864
9.44k
    int code = 0;
865
9.44k
    pdf_dict *d = NULL;
866
9.44k
    uint64_t max_obj = 0;
867
9.44k
    int64_t num, XRefStm = 0;
868
9.44k
    int obj_num;
869
9.44k
    bool known = false;
870
871
9.44k
    if (ctx->repaired)
872
2
        return 0;
873
874
31.8k
    do {
875
31.8k
        uint64_t section_start, section_size;
876
877
31.8k
        code = read_xref_section(ctx, s, &section_start, &section_size);
878
31.8k
        if (code < 0)
879
989
            return code;
880
881
30.8k
        if (section_size > 0 && section_start + section_size - 1 > max_obj)
882
19.8k
            max_obj = section_start + section_size - 1;
883
884
        /* code == 1 => read_xref_section ended with a trailer. */
885
30.8k
    } while (code != 1);
886
887
8.45k
    code = pdfi_read_dict(ctx, ctx->main_stream, 0, 0);
888
8.45k
    if (code < 0)
889
154
        return code;
890
891
8.30k
    d = (pdf_dict *)ctx->stack_top[-1];
892
8.30k
    if (pdfi_type_of(d) != PDF_DICT) {
893
12
        pdfi_pop(ctx, 1);
894
12
        return_error(gs_error_typecheck);
895
12
    }
896
8.29k
    pdfi_countup(d);
897
8.29k
    pdfi_pop(ctx, 1);
898
899
    /* We don't want to pollute the Trailer dictionary with any XRefStm key/value pairs
900
     * which will happen when we do pdfi_merge_dicts(). So we get any XRefStm here and
901
     * if there was one, remove it from the dictionary before we merge with the
902
     * primary trailer.
903
     */
904
8.29k
    code = pdfi_dict_get_int(ctx, d, "XRefStm", &XRefStm);
905
8.29k
    if (code < 0 && code != gs_error_undefined)
906
1
        goto error;
907
908
8.29k
    if (code == 0) {
909
232
        code = pdfi_dict_delete(ctx, d, "XRefStm");
910
232
        if (code < 0)
911
0
            goto error;
912
232
    }
913
914
8.29k
    if (ctx->Trailer == NULL) {
915
7.40k
        ctx->Trailer = d;
916
7.40k
        pdfi_countup(d);
917
7.40k
    } else {
918
890
        code = pdfi_merge_dicts(ctx, ctx->Trailer, d);
919
890
        if (code < 0) {
920
0
            if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREF, "read_xref", "")) < 0) {
921
0
                return code;
922
0
            }
923
0
        }
924
890
    }
925
926
    /* Check if the highest subsection + size exceeds the /Size in the
927
     * trailer dictionary and set a warning flag if it does
928
     */
929
8.29k
    code = pdfi_dict_get_int(ctx, d, "Size", &num);
930
8.29k
    if (code < 0)
931
13
        goto error;
932
933
8.27k
    if (max_obj >= num)
934
531
        pdfi_set_warning(ctx, 0, NULL, W_PDF_BAD_XREF_SIZE, "read_xref", NULL);
935
936
    /* Check if this is a modified file and has any
937
     * previous xref entries.
938
     */
939
8.27k
    code = pdfi_dict_known(ctx, d, "Prev", &known);
940
8.27k
    if (known) {
941
3.41k
        code = pdfi_dict_get_int(ctx, d, "Prev", &num);
942
3.41k
        if (code < 0)
943
12
            goto error;
944
945
3.40k
        if (num < 0 || num > ctx->main_stream_length) {
946
1.21k
            code = gs_note_error(gs_error_rangecheck);
947
1.21k
            goto error;
948
1.21k
        }
949
950
2.18k
        if (pdfi_loop_detector_check_object(ctx, num) == true) {
951
2
            code = gs_note_error(gs_error_circular_reference);
952
2
            goto error;
953
2
        }
954
2.18k
        else {
955
2.18k
            code = pdfi_loop_detector_add_object(ctx, num);
956
2.18k
            if (code < 0)
957
0
                goto error;
958
2.18k
        }
959
960
2.18k
        code = pdfi_seek(ctx, s, num, SEEK_SET);
961
2.18k
        if (code < 0)
962
0
            goto error;
963
964
2.18k
        if (!ctx->repaired) {
965
2.18k
            code = pdfi_read_token(ctx, ctx->main_stream, 0, 0);
966
2.18k
            if (code < 0)
967
87
                goto error;
968
969
2.10k
            if (code == 0) {
970
3
                code = gs_note_error(gs_error_syntaxerror);
971
3
                goto error;
972
3
            }
973
2.10k
        } else {
974
0
            code = 0;
975
0
            goto error;
976
0
        }
977
978
2.09k
        if ((intptr_t)(ctx->stack_top[-1]) == (intptr_t)TOKEN_XREF) {
979
            /* Read old-style xref table */
980
953
            pdfi_pop(ctx, 1);
981
953
            code = read_xref(ctx, ctx->main_stream);
982
953
            if (code < 0)
983
152
                goto error;
984
1.14k
        } else {
985
1.14k
            pdfi_pop(ctx, 1);
986
1.14k
            code = gs_note_error(gs_error_typecheck);
987
1.14k
            goto error;
988
1.14k
        }
989
2.09k
    }
990
991
    /* Now check if this is a hybrid file. */
992
5.66k
    if (XRefStm != 0) {
993
142
        ctx->is_hybrid = true;
994
995
142
        if (ctx->args.pdfdebug)
996
0
            outprintf(ctx->memory, "%% File is a hybrid, containing xref table and xref stream. Reading the stream.\n");
997
998
999
142
        if (pdfi_loop_detector_check_object(ctx, XRefStm) == true) {
1000
0
            code = gs_note_error(gs_error_circular_reference);
1001
0
            goto error;
1002
0
        }
1003
142
        else {
1004
142
            code = pdfi_loop_detector_add_object(ctx, XRefStm);
1005
142
            if (code < 0)
1006
0
                goto error;
1007
142
        }
1008
1009
142
        code = pdfi_loop_detector_mark(ctx);
1010
142
        if (code < 0)
1011
0
            goto error;
1012
1013
        /* Because of the way the code works when we read a file which is a pure
1014
         * xref stream file, we need to read the first integer of 'x y obj'
1015
         * because the xref stream decoding code expects that to be on the stack.
1016
         */
1017
142
        pdfi_seek(ctx, s, XRefStm, SEEK_SET);
1018
1019
142
        code = pdfi_read_bare_int(ctx, ctx->main_stream, &obj_num);
1020
142
        if (code < 0) {
1021
0
            pdfi_set_error(ctx, 0, NULL, E_PDF_BADXREFSTREAM, "read_xref", "");
1022
0
            pdfi_loop_detector_cleartomark(ctx);
1023
0
            goto error;
1024
0
        }
1025
1026
142
        code = pdfi_read_xref_stream_dict(ctx, ctx->main_stream, obj_num);
1027
        /* We could just fall through to the exit here, but choose not to in order to avoid possible mistakes in future */
1028
142
        if (code < 0) {
1029
9
            pdfi_loop_detector_cleartomark(ctx);
1030
9
            goto error;
1031
9
        }
1032
1033
133
        pdfi_loop_detector_cleartomark(ctx);
1034
133
    } else
1035
5.51k
        code = 0;
1036
1037
8.29k
error:
1038
8.29k
    pdfi_countdown(d);
1039
8.29k
    return code;
1040
5.66k
}
1041
1042
int pdfi_read_xref(pdf_context *ctx)
1043
83.3k
{
1044
83.3k
    int code = 0;
1045
83.3k
    int obj_num;
1046
1047
83.3k
    code = pdfi_loop_detector_mark(ctx);
1048
83.3k
    if (code < 0)
1049
0
        return code;
1050
1051
83.3k
    if (ctx->startxref == 0)
1052
47.8k
        goto repair;
1053
1054
35.5k
    code = pdfi_loop_detector_add_object(ctx, ctx->startxref);
1055
35.5k
    if (code < 0)
1056
0
        goto exit;
1057
1058
35.5k
    if (ctx->args.pdfdebug)
1059
0
        outprintf(ctx->memory, "%% Trying to read 'xref' token for xref table, or 'int int obj' for an xref stream\n");
1060
1061
35.5k
    if (ctx->startxref > ctx->main_stream_length - 5) {
1062
8.97k
        if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_rangecheck), NULL, E_PDF_BADSTARTXREF, "pdfi_read_xref", (char *)"startxref offset is beyond end of file")) < 0)
1063
0
            goto exit;
1064
1065
8.97k
        goto repair;
1066
8.97k
    }
1067
26.5k
    if (ctx->startxref < 0) {
1068
319
        if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_rangecheck), NULL, E_PDF_BADSTARTXREF, "pdfi_read_xref", (char *)"startxref offset is before start of file")) < 0)
1069
0
            goto exit;
1070
1071
319
        goto repair;
1072
319
    }
1073
1074
    /* Read the xref(s) */
1075
26.2k
    pdfi_seek(ctx, ctx->main_stream, ctx->startxref, SEEK_SET);
1076
1077
    /* If it starts with an int, it's an xref stream dict */
1078
26.2k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &obj_num);
1079
26.2k
    if (code == 1) {
1080
10.0k
        if (pdfi_check_xref_stream(ctx)) {
1081
7.52k
            code = pdfi_read_xref_stream_dict(ctx, ctx->main_stream, obj_num);
1082
7.52k
            if (code < 0)
1083
2.77k
                goto repair;
1084
7.52k
        } else
1085
2.52k
            goto repair;
1086
16.1k
    } else {
1087
        /* If not, it had better start 'xref', and be an old-style xref table */
1088
16.1k
        code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
1089
16.1k
        if (code != TOKEN_XREF) {
1090
7.72k
            if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_syntaxerror), NULL, E_PDF_BADSTARTXREF, "pdfi_read_xref", (char *)"Failed to read any token at the startxref location")) < 0)
1091
0
                goto exit;
1092
1093
7.72k
            goto repair;
1094
7.72k
        }
1095
1096
8.47k
        code = read_xref(ctx, ctx->main_stream);
1097
8.47k
        if (code < 0)
1098
3.62k
            goto repair;
1099
8.47k
    }
1100
1101
9.58k
    if(ctx->args.pdfdebug && ctx->xref_table) {
1102
0
        int i, j;
1103
0
        xref_entry *entry;
1104
0
        char Buffer[32];
1105
1106
0
        outprintf(ctx->memory, "\n%% Dumping xref table\n");
1107
0
        for (i=0;i < ctx->xref_table->xref_size;i++) {
1108
0
            entry = &ctx->xref_table->xref[i];
1109
0
            if(entry->compressed) {
1110
0
                outprintf(ctx->memory, "*");
1111
0
                gs_snprintf(Buffer, sizeof(Buffer), "%"PRId64"", entry->object_num);
1112
0
                j = 10 - strlen(Buffer);
1113
0
                while(j--) {
1114
0
                    outprintf(ctx->memory, " ");
1115
0
                }
1116
0
                outprintf(ctx->memory, "%s ", Buffer);
1117
1118
0
                gs_snprintf(Buffer, sizeof(Buffer), "%ld", entry->u.compressed.compressed_stream_num);
1119
0
                j = 10 - strlen(Buffer);
1120
0
                while(j--) {
1121
0
                    outprintf(ctx->memory, " ");
1122
0
                }
1123
0
                outprintf(ctx->memory, "%s ", Buffer);
1124
1125
0
                gs_snprintf(Buffer, sizeof(Buffer), "%ld", entry->u.compressed.object_index);
1126
0
                j = 10 - strlen(Buffer);
1127
0
                while(j--) {
1128
0
                    outprintf(ctx->memory, " ");
1129
0
                }
1130
0
                outprintf(ctx->memory, "%s ", Buffer);
1131
0
            }
1132
0
            else {
1133
0
                outprintf(ctx->memory, " ");
1134
1135
0
                gs_snprintf(Buffer, sizeof(Buffer), "%ld", entry->object_num);
1136
0
                j = 10 - strlen(Buffer);
1137
0
                while(j--) {
1138
0
                    outprintf(ctx->memory, " ");
1139
0
                }
1140
0
                outprintf(ctx->memory, "%s ", Buffer);
1141
1142
0
                gs_snprintf(Buffer, sizeof(Buffer), "%"PRIdOFFSET"", entry->u.uncompressed.offset);
1143
0
                j = 10 - strlen(Buffer);
1144
0
                while(j--) {
1145
0
                    outprintf(ctx->memory, " ");
1146
0
                }
1147
0
                outprintf(ctx->memory, "%s ", Buffer);
1148
1149
0
                gs_snprintf(Buffer, sizeof(Buffer), "%ld", entry->u.uncompressed.generation_num);
1150
0
                j = 10 - strlen(Buffer);
1151
0
                while(j--) {
1152
0
                    outprintf(ctx->memory, " ");
1153
0
                }
1154
0
                outprintf(ctx->memory, "%s ", Buffer);
1155
0
            }
1156
0
            if (entry->free)
1157
0
                outprintf(ctx->memory, "f\n");
1158
0
            else
1159
0
                outprintf(ctx->memory, "n\n");
1160
0
        }
1161
0
    }
1162
9.58k
    if (ctx->args.pdfdebug)
1163
0
        outprintf(ctx->memory, "\n");
1164
1165
9.58k
 exit:
1166
9.58k
    (void)pdfi_loop_detector_cleartomark(ctx);
1167
1168
9.58k
    if (code < 0)
1169
0
        return code;
1170
1171
9.58k
    return 0;
1172
1173
73.7k
repair:
1174
73.7k
    (void)pdfi_loop_detector_cleartomark(ctx);
1175
73.7k
    if (!ctx->repaired && !ctx->args.pdfstoponerror)
1176
73.7k
        return(pdfi_repair_file(ctx));
1177
51
    return 0;
1178
73.7k
}