Coverage Report

Created: 2026-08-08 08:00

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/ghostpdl/pdf/pdf_xref.c
Line
Count
Source
1
/* Copyright (C) 2018-2025 Artifex Software, Inc.
2
   All Rights Reserved.
3
4
   This software is provided AS-IS with no warranty, either express or
5
   implied.
6
7
   This software is distributed under license and may not be copied,
8
   modified or distributed except as expressly authorized under the terms
9
   of the license contained in the file LICENSE in this distribution.
10
11
   Refer to licensing information at http://www.artifex.com or contact
12
   Artifex Software, Inc.,  39 Mesa Street, Suite 108A, San Francisco,
13
   CA 94129, USA, for further information.
14
*/
15
16
/* xref parsing */
17
18
#include "pdf_int.h"
19
#include "pdf_stack.h"
20
#include "pdf_xref.h"
21
#include "pdf_file.h"
22
#include "pdf_loop_detect.h"
23
#include "pdf_dict.h"
24
#include "pdf_array.h"
25
#include "pdf_repair.h"
26
27
static int resize_xref(pdf_context *ctx, uint64_t new_size)
28
17.4k
{
29
17.4k
    xref_entry *new_xrefs;
30
31
    /* Although we can technically handle object numbers larger than this, on some systems (32-bit Windows)
32
     * memset is limited to a (signed!) integer for the size of memory to clear. We could deal
33
     * with this by clearing the memory in blocks, but really, this is almost certainly a
34
     * corrupted file or something.
35
     */
36
17.4k
    if (new_size >= (0x7ffffff / sizeof(xref_entry)))
37
22
        return_error(gs_error_rangecheck);
38
39
17.4k
    new_xrefs = (xref_entry *)gs_alloc_bytes(ctx->memory, (size_t)(new_size) * sizeof(xref_entry), "read_xref_stream allocate xref table entries");
40
17.4k
    if (new_xrefs == NULL){
41
0
        pdfi_countdown(ctx->xref_table);
42
0
        ctx->xref_table = NULL;
43
0
        return_error(gs_error_VMerror);
44
0
    }
45
17.4k
    memset(new_xrefs, 0x00, (new_size) * sizeof(xref_entry));
46
17.4k
    memcpy(new_xrefs, ctx->xref_table->xref, ctx->xref_table->xref_size * sizeof(xref_entry));
47
17.4k
    gs_free_object(ctx->memory, ctx->xref_table->xref, "reallocated xref entries");
48
17.4k
    ctx->xref_table->xref = new_xrefs;
49
17.4k
    ctx->xref_table->xref_size = new_size;
50
17.4k
    return 0;
51
17.4k
}
52
53
static int read_xref_stream_entries(pdf_context *ctx, pdf_c_stream *s, int64_t first, int64_t last, int64_t *W)
54
14.2k
{
55
14.2k
    uint i, j;
56
14.2k
    uint64_t field_width = 0;
57
14.2k
    uint32_t type = 0;
58
14.2k
    uint64_t objnum = 0, gen = 0;
59
14.2k
    byte *Buffer;
60
14.2k
    int64_t bytes = 0;
61
14.2k
    xref_entry *entry;
62
63
    /* Find max number of bytes to be read */
64
14.2k
    field_width = W[0];
65
14.2k
    if (W[1] > field_width)
66
14.1k
        field_width = W[1];
67
14.2k
    if (W[2] > field_width)
68
18
        field_width = W[2];
69
70
14.2k
    Buffer = gs_alloc_bytes(ctx->memory, field_width, "read_xref_stream_entry working buffer");
71
14.2k
    if (Buffer == NULL)
72
0
        return_error(gs_error_VMerror);
73
74
690k
    for (i=first;i<=last; i++){
75
        /* Defaults if W[n] = 0 */
76
676k
        type = 1;
77
676k
        objnum = gen = 0;
78
79
676k
        if (W[0] != 0) {
80
673k
            type = 0;
81
673k
            bytes = pdfi_read_bytes(ctx, Buffer, 1, W[0], s);
82
673k
            if (bytes < W[0]){
83
114
                gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry, free working buffer (error)");
84
114
                return_error(gs_error_ioerror);
85
114
            }
86
1.34M
            for (j=0;j<W[0];j++)
87
673k
                type = (type << 8) + Buffer[j];
88
673k
        }
89
90
676k
        if (W[1] != 0) {
91
676k
            bytes = pdfi_read_bytes(ctx, Buffer, 1, W[1], s);
92
676k
            if (bytes < W[1]){
93
28
                gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry free working buffer (error)");
94
28
                return_error(gs_error_ioerror);
95
28
            }
96
2.37M
            for (j=0;j<W[1];j++)
97
1.69M
                objnum = (objnum << 8) + Buffer[j];
98
676k
        }
99
100
676k
        if (W[2] != 0) {
101
663k
            bytes = pdfi_read_bytes(ctx, Buffer, 1, W[2], s);
102
663k
            if (bytes < W[2]){
103
33
                gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry, free working buffer (error)");
104
33
                return_error(gs_error_ioerror);
105
33
            }
106
1.34M
            for (j=0;j<W[2];j++)
107
680k
                gen = (gen << 8) + Buffer[j];
108
663k
        }
109
110
676k
        entry = &ctx->xref_table->xref[i];
111
676k
        if (entry->object_num != 0 && !entry->free)
112
3.93k
            continue;
113
114
672k
        entry->compressed = false;
115
672k
        entry->free = false;
116
672k
        entry->object_num = i;
117
672k
        entry->cache = NULL;
118
119
672k
        switch(type) {
120
14.8k
            case 0:
121
14.8k
                entry->free = true;
122
14.8k
                entry->u.uncompressed.offset = objnum;         /* For free objects we use the offset to store the object number of the next free object */
123
14.8k
                entry->u.uncompressed.generation_num = gen;    /* And the generation number is the numebr to use if this object is used again */
124
14.8k
                break;
125
222k
            case 1:
126
222k
                entry->u.uncompressed.offset = objnum;
127
222k
                entry->u.uncompressed.generation_num = gen;
128
222k
                break;
129
435k
            case 2:
130
435k
                entry->compressed = true;
131
435k
                entry->u.compressed.compressed_stream_num = objnum;   /* The object number of the compressed stream */
132
435k
                entry->u.compressed.object_index = gen;               /* And the index of the object within the stream */
133
435k
                break;
134
126
            default:
135
126
                gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry, free working buffer");
136
126
                return_error(gs_error_rangecheck);
137
0
                break;
138
672k
        }
139
672k
    }
140
13.9k
    gs_free_object(ctx->memory, Buffer, "read_xref_stream_entry, free working buffer");
141
13.9k
    return 0;
142
14.2k
}
143
144
/* Forward definition */
145
static int read_xref(pdf_context *ctx, pdf_c_stream *s);
146
static int pdfi_check_xref_stream(pdf_context *ctx);
147
/* These two routines are recursive.... */
148
static int pdfi_read_xref_stream_dict(pdf_context *ctx, pdf_c_stream *s, int obj_num);
149
150
static int pdfi_process_xref_stream(pdf_context *ctx, pdf_stream *stream_obj, pdf_c_stream *s)
151
11.1k
{
152
11.1k
    pdf_c_stream *XRefStrm;
153
11.1k
    int code, i;
154
11.1k
    pdf_dict *sdict = NULL;
155
11.1k
    pdf_name *n;
156
11.1k
    pdf_array *a;
157
11.1k
    int64_t size;
158
11.1k
    int64_t num;
159
11.1k
    int64_t W[3] = {0, 0, 0};
160
11.1k
    int objnum;
161
11.1k
    bool known = false;
162
163
11.1k
    if (pdfi_type_of(stream_obj) != PDF_STREAM)
164
0
        return_error(gs_error_typecheck);
165
166
11.1k
    code = pdfi_dict_from_obj(ctx, (pdf_obj *)stream_obj, &sdict);
167
11.1k
    if (code < 0)
168
0
        return code;
169
170
11.1k
    code = pdfi_dict_get_type(ctx, sdict, "Type", PDF_NAME, (pdf_obj **)&n);
171
11.1k
    if (code < 0)
172
46
        return code;
173
174
11.1k
    if (n->length != 4 || memcmp(n->data, "XRef", 4) != 0) {
175
23
        pdfi_countdown(n);
176
23
        return_error(gs_error_syntaxerror);
177
23
    }
178
11.1k
    pdfi_countdown(n);
179
180
11.1k
    code = pdfi_dict_get_int(ctx, sdict, "Size", &size);
181
11.1k
    if (code < 0)
182
10
        return code;
183
11.0k
    if (size < 1)
184
9
        return 0;
185
186
11.0k
    if (size < 0 || size > floor((double)ARCH_MAX_SIZE_T / (double)sizeof(xref_entry)))
187
0
        return_error(gs_error_rangecheck);
188
189
    /* If this is the first xref stream then allocate the xref table and store the trailer */
190
11.0k
    if (ctx->xref_table == NULL) {
191
7.04k
        ctx->xref_table = (xref_table_t *)gs_alloc_bytes(ctx->memory, sizeof(xref_table_t), "read_xref_stream allocate xref table");
192
7.04k
        if (ctx->xref_table == NULL) {
193
0
            return_error(gs_error_VMerror);
194
0
        }
195
7.04k
        memset(ctx->xref_table, 0x00, sizeof(xref_table_t));
196
7.04k
        ctx->xref_table->xref = (xref_entry *)gs_alloc_bytes(ctx->memory, (size_t)size * sizeof(xref_entry), "read_xref_stream allocate xref table entries");
197
7.04k
        if (ctx->xref_table->xref == NULL){
198
2
            gs_free_object(ctx->memory, ctx->xref_table, "failed to allocate xref table entries");
199
2
            ctx->xref_table = NULL;
200
2
            return_error(gs_error_VMerror);
201
2
        }
202
7.03k
        memset(ctx->xref_table->xref, 0x00, size * sizeof(xref_entry));
203
7.03k
        ctx->xref_table->ctx = ctx;
204
7.03k
        ctx->xref_table->type = PDF_XREF_TABLE;
205
7.03k
        ctx->xref_table->xref_size = size;
206
#if REFCNT_DEBUG
207
        ctx->xref_table->UID = ctx->ref_UID++;
208
        outprintf(ctx->memory, "Allocated xref table with UID %"PRIi64"\n", ctx->xref_table->UID);
209
#endif
210
7.03k
        pdfi_countup(ctx->xref_table);
211
212
7.03k
        pdfi_countdown(ctx->Trailer);
213
214
7.03k
        ctx->Trailer = sdict;
215
7.03k
        pdfi_countup(sdict);
216
7.03k
    } else {
217
4.04k
        if (size > ctx->xref_table->xref_size)
218
3
            return_error(gs_error_rangecheck);
219
220
4.03k
        code = pdfi_merge_dicts(ctx, ctx->Trailer, sdict);
221
4.03k
        if (code < 0 && (code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREF, "pdfi_process_xref_stream", NULL)) < 0) {
222
0
            goto exit;
223
0
        }
224
4.03k
    }
225
226
11.0k
    pdfi_seek(ctx, ctx->main_stream, pdfi_stream_offset(ctx, stream_obj), SEEK_SET);
227
228
    /* Bug #691220 has a PDF file with a compressed XRef, the stream dictionary has
229
     * a /DecodeParms entry for the stream, which has a /Colors value of 5, which makes
230
     * *no* sense whatever. If we try to apply a Predictor then we end up in a loop trying
231
     * to read 5 colour samples. Rather than meddles with more parameters to the filter
232
     * code, we'll just remove the Colors entry from the DecodeParms dictionary,
233
     * because it is nonsense. This means we'll get the (sensible) default value of 1.
234
     */
235
11.0k
    code = pdfi_dict_known(ctx, sdict, "DecodeParms", &known);
236
11.0k
    if (code < 0)
237
0
        return code;
238
239
11.0k
    if (known) {
240
10.1k
        pdf_dict *DP;
241
10.1k
        double f;
242
10.1k
        pdf_obj *name;
243
244
10.1k
        code = pdfi_dict_get_type(ctx, sdict, "DecodeParms", PDF_DICT, (pdf_obj **)&DP);
245
10.1k
        if (code < 0)
246
0
            return code;
247
248
10.1k
        code = pdfi_dict_knownget_number(ctx, DP, "Colors", &f);
249
10.1k
        if (code < 0) {
250
0
            pdfi_countdown(DP);
251
0
            return code;
252
0
        }
253
10.1k
        if (code > 0 && f != (double)1)
254
0
        {
255
0
            code = pdfi_name_alloc(ctx, (byte *)"Colors", 6, &name);
256
0
            if (code < 0) {
257
0
                pdfi_countdown(DP);
258
0
                return code;
259
0
            }
260
0
            pdfi_countup(name);
261
262
0
            code = pdfi_dict_delete_pair(ctx, DP, (pdf_name *)name);
263
0
            pdfi_countdown(name);
264
0
            if (code < 0) {
265
0
                pdfi_countdown(DP);
266
0
                return code;
267
0
            }
268
0
        }
269
10.1k
        pdfi_countdown(DP);
270
10.1k
    }
271
272
11.0k
    code = pdfi_filter_no_decryption(ctx, stream_obj, s, &XRefStrm, false);
273
11.0k
    if (code < 0) {
274
43
        pdfi_countdown(ctx->xref_table);
275
43
        ctx->xref_table = NULL;
276
43
        return code;
277
43
    }
278
279
11.0k
    code = pdfi_dict_get_type(ctx, sdict, "W", PDF_ARRAY, (pdf_obj **)&a);
280
11.0k
    if (code < 0) {
281
7
        pdfi_close_file(ctx, XRefStrm);
282
7
        pdfi_countdown(ctx->xref_table);
283
7
        ctx->xref_table = NULL;
284
7
        return code;
285
7
    }
286
287
11.0k
    if (pdfi_array_size(a) != 3) {
288
11
        pdfi_countdown(a);
289
11
        pdfi_close_file(ctx, XRefStrm);
290
11
        pdfi_countdown(ctx->xref_table);
291
11
        ctx->xref_table = NULL;
292
11
        return_error(gs_error_rangecheck);
293
11
    }
294
43.9k
    for (i=0;i<3;i++) {
295
33.0k
        code = pdfi_array_get_int(ctx, a, (uint64_t)i, (int64_t *)&W[i]);
296
33.0k
        if (code < 0 || W[i] < 0) {
297
34
            pdfi_countdown(a);
298
34
            pdfi_close_file(ctx, XRefStrm);
299
34
            pdfi_countdown(ctx->xref_table);
300
34
            ctx->xref_table = NULL;
301
34
            if (W[i] < 0)
302
11
                code = gs_note_error(gs_error_rangecheck);
303
34
            return code;
304
34
        }
305
33.0k
    }
306
10.9k
    pdfi_countdown(a);
307
308
    /* W[0] is either:
309
     * 0 (no type field) or a single byte with the type.
310
     * W[1] is either:
311
     * The object number of the next free object, the byte offset of this object in the file or the object5 number of the object stream where this object is stored.
312
     * W[2] is either:
313
     * The generation number to use if this object is used again, the generation number of the object or the index of this object within the object stream.
314
     *
315
     * Object and generation numbers are limited to unsigned 64-bit values, as are bytes offsets in the file, indexes of objects within the stream likewise (actually
316
     * most of these are generally 32-bit max). So we can limit the field widths to 8 bytes, enough to hold a 64-bit number.
317
     * Even if a later version of the spec makes these larger (which seems unlikely!) we still cna't cope with integers > 64-bits.
318
     */
319
10.9k
    if (W[0] > 1 || W[1] > 8 || W[2] > 8) {
320
28
        pdfi_close_file(ctx, XRefStrm);
321
28
        pdfi_countdown(ctx->xref_table);
322
28
        ctx->xref_table = NULL;
323
28
        return code;
324
28
    }
325
326
10.9k
    code = pdfi_dict_get_type(ctx, sdict, "Index", PDF_ARRAY, (pdf_obj **)&a);
327
10.9k
    if (code == gs_error_undefined) {
328
3.84k
        code = read_xref_stream_entries(ctx, XRefStrm, 0, size - 1, W);
329
3.84k
        if (code < 0) {
330
93
            pdfi_close_file(ctx, XRefStrm);
331
93
            pdfi_countdown(ctx->xref_table);
332
93
            ctx->xref_table = NULL;
333
93
            return code;
334
93
        }
335
7.10k
    } else {
336
7.10k
        int64_t start, size;
337
338
7.10k
        if (code < 0) {
339
3
            pdfi_close_file(ctx, XRefStrm);
340
3
            pdfi_countdown(ctx->xref_table);
341
3
            ctx->xref_table = NULL;
342
3
            return code;
343
3
        }
344
345
7.10k
        if (pdfi_array_size(a) & 1) {
346
12
            pdfi_countdown(a);
347
12
            pdfi_close_file(ctx, XRefStrm);
348
12
            pdfi_countdown(ctx->xref_table);
349
12
            ctx->xref_table = NULL;
350
12
            return_error(gs_error_rangecheck);
351
12
        }
352
353
17.2k
        for (i=0;i < pdfi_array_size(a);i+=2){
354
10.4k
            code = pdfi_array_get_int(ctx, a, (uint64_t)i, &start);
355
10.4k
            if (code < 0 || start < 0) {
356
15
                pdfi_countdown(a);
357
15
                pdfi_close_file(ctx, XRefStrm);
358
15
                pdfi_countdown(ctx->xref_table);
359
15
                ctx->xref_table = NULL;
360
15
                return code;
361
15
            }
362
363
10.4k
            code = pdfi_array_get_int(ctx, a, (uint64_t)i+1, &size);
364
10.4k
            if (code < 0) {
365
12
                pdfi_countdown(a);
366
12
                pdfi_close_file(ctx, XRefStrm);
367
12
                pdfi_countdown(ctx->xref_table);
368
12
                ctx->xref_table = NULL;
369
12
                return code;
370
12
            }
371
372
10.4k
            if (size < 1)
373
11
                continue;
374
375
10.4k
            if (start + size >= ctx->xref_table->xref_size) {
376
6.42k
                code = resize_xref(ctx, start + size);
377
6.42k
                if (code < 0) {
378
6
                    pdfi_countdown(a);
379
6
                    pdfi_close_file(ctx, XRefStrm);
380
6
                    pdfi_countdown(ctx->xref_table);
381
6
                    ctx->xref_table = NULL;
382
6
                    return code;
383
6
                }
384
6.42k
            }
385
386
10.4k
            code = read_xref_stream_entries(ctx, XRefStrm, start, start + size - 1, W);
387
10.4k
            if (code < 0) {
388
208
                pdfi_countdown(a);
389
208
                pdfi_close_file(ctx, XRefStrm);
390
208
                pdfi_countdown(ctx->xref_table);
391
208
                ctx->xref_table = NULL;
392
208
                return code;
393
208
            }
394
10.4k
        }
395
7.09k
    }
396
10.6k
    pdfi_countdown(a);
397
398
10.6k
    pdfi_close_file(ctx, XRefStrm);
399
400
10.6k
    code = pdfi_dict_get_int(ctx, sdict, "Prev", &num);
401
10.6k
    if (code == gs_error_undefined)
402
4.26k
        return 0;
403
404
6.34k
    if (code < 0)
405
12
        return code;
406
407
6.33k
    if (num < 0 || num > ctx->main_stream_length)
408
1.93k
        return_error(gs_error_rangecheck);
409
410
4.39k
    if (pdfi_loop_detector_check_object(ctx, num) == true)
411
14
        return_error(gs_error_circular_reference);
412
4.37k
    else {
413
4.37k
        code = pdfi_loop_detector_add_object(ctx, num);
414
4.37k
        if (code < 0)
415
0
            return code;
416
4.37k
    }
417
418
4.37k
    if(ctx->args.pdfdebug)
419
0
        outprintf(ctx->memory, "%% Reading /Prev xref\n");
420
421
4.37k
    pdfi_seek(ctx, s, num, SEEK_SET);
422
423
4.37k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &objnum);
424
4.37k
    if (code == 1) {
425
4.07k
        if (pdfi_check_xref_stream(ctx))
426
4.04k
            return pdfi_read_xref_stream_dict(ctx, s, objnum);
427
4.07k
    }
428
429
338
    code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
430
338
    if (code < 0)
431
0
        return code;
432
338
    if (code == TOKEN_XREF) {
433
35
        if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_syntaxerror), NULL, E_PDF_PREV_NOT_XREF_STREAM, "pdfi_process_xref_stream", NULL)) < 0) {
434
0
            goto exit;
435
0
        }
436
        /* Read old-style xref table */
437
35
        return(read_xref(ctx, ctx->main_stream));
438
35
    }
439
303
exit:
440
303
    return_error(gs_error_syntaxerror);
441
338
}
442
443
static int pdfi_check_xref_stream(pdf_context *ctx)
444
14.1k
{
445
14.1k
    gs_offset_t offset;
446
14.1k
    int gen_num, code = 0;
447
448
14.1k
    offset = pdfi_unread_tell(ctx);
449
450
14.1k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &gen_num);
451
14.1k
    if (code <= 0) {
452
1.03k
        code = 0;
453
1.03k
        goto exit;
454
1.03k
    }
455
456
    /* Try to read 'obj' */
457
13.1k
    code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
458
13.1k
    if (code <= 0) {
459
0
        code = 0;
460
0
        goto exit;
461
0
    }
462
463
    /* Third element must be obj, or it's not a valid xref */
464
13.1k
    if (code != TOKEN_OBJ)
465
1.52k
        code = 0;
466
11.5k
    else
467
11.5k
        code = 1;
468
469
14.1k
exit:
470
14.1k
    pdfi_seek(ctx, ctx->main_stream, offset, SEEK_SET);
471
14.1k
    return code;
472
13.1k
}
473
474
static int pdfi_read_xref_stream_dict(pdf_context *ctx, pdf_c_stream *s, int obj_num)
475
11.7k
{
476
11.7k
    int code;
477
11.7k
    int gen_num;
478
479
11.7k
    if (ctx->args.pdfdebug)
480
0
        outprintf(ctx->memory, "\n%% Reading PDF 1.5+ xref stream\n");
481
482
    /* We have the obj_num. Lets try for obj_num gen obj as a XRef stream */
483
11.7k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &gen_num);
484
11.7k
    if (code <= 0) {
485
0
        if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREFSTREAM, "pdfi_read_xref_stream_dict", "")) < 0) {
486
0
            return code;
487
0
        }
488
0
        return(pdfi_repair_file(ctx));
489
0
    }
490
491
    /* Try to read 'obj' */
492
11.7k
    code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
493
11.7k
    if (code < 0)
494
0
        return code;
495
11.7k
    if (code == 0)
496
0
        return_error(gs_error_syntaxerror);
497
498
    /* Third element must be obj, or it's not a valid xref */
499
11.7k
    if (code != TOKEN_OBJ) {
500
0
        if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_rangecheck), NULL, E_PDF_BAD_XREFSTMOFFSET, "pdfi_read_xref_stream_dict", "")) < 0) {
501
0
            return code;
502
0
        }
503
0
        return(pdfi_repair_file(ctx));
504
0
    }
505
506
482k
    do {
507
482k
        code = pdfi_read_token(ctx, ctx->main_stream, obj_num, gen_num);
508
482k
        if (code <= 0) {
509
430
            if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREFSTREAM, "pdfi_read_xref_stream_dict", NULL)) < 0) {
510
0
                return code;
511
0
            }
512
430
            return pdfi_repair_file(ctx);
513
430
        }
514
515
482k
        if (pdfi_count_stack(ctx) >= 2 && pdfi_type_of(ctx->stack_top[-1]) == PDF_FAST_KEYWORD) {
516
13.0k
            uintptr_t keyword = (uintptr_t)ctx->stack_top[-1];
517
13.0k
            if (keyword == TOKEN_STREAM) {
518
11.1k
                pdf_dict *dict;
519
11.1k
                pdf_stream *sdict = NULL;
520
11.1k
                int64_t Length;
521
522
                /* Remove the 'stream' token from the stack, should leave a dictionary object on the stack */
523
11.1k
                pdfi_pop(ctx, 1);
524
11.1k
                if (pdfi_type_of(ctx->stack_top[-1]) != PDF_DICT) {
525
21
                    if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREFSTREAM, "pdfi_read_xref_stream_dict", NULL)) < 0) {
526
0
                        return code;
527
0
                    }
528
21
                    return pdfi_repair_file(ctx);
529
21
                }
530
11.1k
                dict = (pdf_dict *)ctx->stack_top[-1];
531
532
                /* Convert the dict into a stream (sdict comes back with at least one ref) */
533
11.1k
                code = pdfi_obj_dict_to_stream(ctx, dict, &sdict, true);
534
                /* Pop off the dict */
535
11.1k
                pdfi_pop(ctx, 1);
536
11.1k
                if (code < 0) {
537
0
                    if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREFSTREAM, "pdfi_read_xref_stream_dict", NULL)) < 0) {
538
0
                        return code;
539
0
                    }
540
                    /* TODO: should I return code instead of trying to repair?
541
                     * Normally the above routine should not fail so something is
542
                     * probably seriously fubar.
543
                     */
544
0
                    return pdfi_repair_file(ctx);
545
0
                }
546
11.1k
                dict = NULL;
547
548
                /* Init the stuff for the stream */
549
11.1k
                sdict->stream_offset = pdfi_unread_tell(ctx);
550
11.1k
                sdict->object_num = obj_num;
551
11.1k
                sdict->generation_num = gen_num;
552
553
11.1k
                code = pdfi_dict_get_int(ctx, sdict->stream_dict, "Length", &Length);
554
11.1k
                if (code < 0) {
555
                    /* TODO: Not positive this will actually have a length -- just use 0 */
556
43
                    (void)pdfi_set_error_var(ctx, 0, NULL, E_PDF_BADSTREAM, "pdfi_read_xref_stream_dict", "Xref Stream object %u missing mandatory keyword /Length\n", obj_num);
557
43
                    code = 0;
558
43
                    Length = 0;
559
43
                }
560
11.1k
                sdict->Length = Length;
561
11.1k
                sdict->length_valid = true;
562
563
11.1k
                code = pdfi_process_xref_stream(ctx, sdict, ctx->main_stream);
564
11.1k
                pdfi_countdown(sdict);
565
11.1k
                if (code < 0) {
566
3.00k
                    pdfi_set_error(ctx, gs_note_error(gs_error_syntaxerror), NULL, E_PDF_PREV_NOT_XREF_STREAM, "pdfi_read_xref_stream_dict", NULL);
567
3.00k
                    return code;
568
3.00k
                }
569
8.16k
                break;
570
11.1k
            } else if (keyword == TOKEN_ENDOBJ) {
571
                /* Something went wrong, this is not a stream dictionary */
572
124
                if ((code = pdfi_set_error_var(ctx, 0, NULL, E_PDF_BADSTREAM, "pdfi_read_xref_stream_dict", "Xref Stream object %u missing mandatory keyword /Length\n", obj_num)) < 0) {
573
0
                    return code;
574
0
                }
575
124
                return(pdfi_repair_file(ctx));
576
124
            }
577
13.0k
        }
578
482k
    } while(1);
579
8.16k
    return 0;
580
11.7k
}
581
582
static int skip_to_digit(pdf_context *ctx, pdf_c_stream *s, unsigned int limit)
583
2.69k
{
584
2.69k
    int c, read = 0;
585
586
10.0k
    do {
587
10.0k
        c = pdfi_read_byte(ctx, s);
588
10.0k
        if (c < 0)
589
0
            return_error(gs_error_ioerror);
590
10.0k
        if (c >= '0' && c <= '9') {
591
2.42k
            pdfi_unread_byte(ctx, s, (byte)c);
592
2.42k
            return read;
593
2.42k
        }
594
7.63k
        read++;
595
7.63k
    } while (read < limit);
596
597
267
    return read;
598
2.69k
}
599
600
static int read_digits(pdf_context *ctx, pdf_c_stream *s, byte *Buffer, int limit)
601
2.69k
{
602
2.69k
    int c, read = 0;
603
604
    /* Since the "limit" is a value calculated by the caller,
605
       it's easier to check it in one place (here) than before
606
       every call.
607
     */
608
2.69k
    if (limit <= 0)
609
274
        return_error(gs_error_syntaxerror);
610
611
    /* We assume that Buffer always has limit+1 bytes available, so we can
612
     * safely terminate it. */
613
614
14.5k
    do {
615
14.5k
        c = pdfi_read_byte(ctx, s);
616
14.5k
        if (c < 0)
617
0
            return_error(gs_error_ioerror);
618
14.5k
        if (c < '0' || c > '9') {
619
1.05k
            pdfi_unread_byte(ctx, s, c);
620
1.05k
            break;
621
1.05k
        }
622
13.4k
        *Buffer++ = (byte)c;
623
13.4k
        read++;
624
13.4k
    } while (read < limit);
625
2.41k
    *Buffer = 0;
626
627
2.41k
    return read;
628
2.41k
}
629
630
631
static int read_xref_entry_slow(pdf_context *ctx, pdf_c_stream *s, gs_offset_t *offset, uint32_t *generation_num, unsigned char *free)
632
1.36k
{
633
1.36k
    byte Buffer[20];
634
1.36k
    int c, code, read = 0;
635
636
    /* First off, find a number. If we don't find one, and read 20 bytes, throw an error */
637
1.36k
    code = skip_to_digit(ctx, s, 20);
638
1.36k
    if (code < 0)
639
0
        return code;
640
1.36k
    read += code;
641
642
    /* Now read a number */
643
1.36k
    code = read_digits(ctx, s, (byte *)&Buffer,  (read > 10 ? 20 - read : 10));
644
1.36k
    if (code < 0)
645
41
        return code;
646
1.32k
    read += code;
647
648
1.32k
    *offset = atol((const char *)Buffer);
649
650
    /* find next number */
651
1.32k
    code = skip_to_digit(ctx, s, 20 - read);
652
1.32k
    if (code < 0)
653
0
        return code;
654
1.32k
    read += code;
655
656
    /* and read it */
657
1.32k
    code = read_digits(ctx, s, (byte *)&Buffer, (read > 15 ? 20 - read : 5));
658
1.32k
    if (code < 0)
659
233
        return code;
660
1.09k
    read += code;
661
662
1.09k
    *generation_num = atol((const char *)Buffer);
663
664
1.89k
    do {
665
1.89k
        c = pdfi_read_byte(ctx, s);
666
1.89k
        if (c < 0)
667
0
            return_error(gs_error_ioerror);
668
1.89k
        read ++;
669
1.89k
        if (c == 0x09 || c == 0x20)
670
827
            continue;
671
1.07k
        if (c == 'n' || c == 'f') {
672
616
            *free = (unsigned char)c;
673
616
            break;
674
616
        } else {
675
455
            return_error(gs_error_syntaxerror);
676
455
        }
677
1.07k
    } while (read < 20);
678
638
    if (read >= 20)
679
35
        return_error(gs_error_syntaxerror);
680
681
1.55k
    do {
682
1.55k
        c = pdfi_read_byte(ctx, s);
683
1.55k
        if (c < 0)
684
0
            return_error(gs_error_syntaxerror);
685
1.55k
        read++;
686
1.55k
        if (c == 0x20 || c == 0x09 || c == 0x0d || c == 0x0a)
687
754
            continue;
688
1.55k
    } while (read < 20);
689
603
    return 0;
690
603
}
691
692
static int write_offset(byte *B, gs_offset_t o, unsigned int g, unsigned char free)
693
603
{
694
603
    byte b[20], *ptr = B;
695
603
    int index = 0;
696
697
603
    gs_snprintf((char *)b, sizeof(b), "%"PRIdOFFSET"", o);
698
603
    if (strlen((const char *)b) > 10)
699
0
        return_error(gs_error_rangecheck);
700
4.94k
    for(index=0;index < 10 - strlen((const char *)b); index++) {
701
4.34k
        *ptr++ = 0x30;
702
4.34k
    }
703
603
    memcpy(ptr, b, strlen((const char *)b));
704
603
    ptr += strlen((const char *)b);
705
603
    *ptr++ = 0x20;
706
707
603
    gs_snprintf((char *)b, sizeof(b), "%d", g);
708
603
    if (strlen((const char *)b) > 5)
709
0
        return_error(gs_error_rangecheck);
710
2.75k
    for(index=0;index < 5 - strlen((const char *)b);index++) {
711
2.15k
        *ptr++ = 0x30;
712
2.15k
    }
713
603
    memcpy(ptr, b, strlen((const char *)b));
714
603
    ptr += strlen((const char *)b);
715
603
    *ptr++ = 0x20;
716
603
    *ptr++ = free;
717
603
    *ptr++ = 0x20;
718
603
    *ptr++ = 0x0d;
719
603
    return 0;
720
603
}
721
722
static int read_xref_section(pdf_context *ctx, pdf_c_stream *s, uint64_t *section_start, uint64_t *section_size)
723
32.5k
{
724
32.5k
    int code = 0, i, j;
725
32.5k
    int start = 0;
726
32.5k
    int size = 0;
727
32.5k
    int64_t bytes = 0;
728
32.5k
    char Buffer[21];
729
730
32.5k
    *section_start = *section_size = 0;
731
732
32.5k
    if (ctx->args.pdfdebug)
733
0
        outprintf(ctx->memory, "\n%% Reading xref section\n");
734
735
32.5k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &start);
736
32.5k
    if (code < 0) {
737
        /* Not an int, might be a keyword */
738
8.62k
        code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
739
8.62k
        if (code < 0)
740
0
            return code;
741
742
8.62k
        if (code != TOKEN_TRAILER) {
743
            /* element is not an integer, and not a keyword - not a valid xref */
744
125
            return_error(gs_error_typecheck);
745
125
        }
746
8.49k
        return 1;
747
8.62k
    }
748
749
23.9k
    if (start < 0)
750
13
        return_error(gs_error_rangecheck);
751
752
23.9k
    *section_start = start;
753
754
23.9k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &size);
755
23.9k
    if (code < 0)
756
22
        return code;
757
23.8k
    if (code == 0)
758
38
        return_error(gs_error_syntaxerror);
759
760
    /* Zero sized xref sections are valid; see the file attached to
761
     * bug 704947 for an example. */
762
23.8k
    if (size < 0)
763
11
        return_error(gs_error_rangecheck);
764
765
23.8k
    *section_size = size;
766
767
23.8k
    if (ctx->args.pdfdebug)
768
0
        outprintf(ctx->memory, "\n%% Section starts at %d and has %d entries\n", (unsigned int) start, (unsigned int)size);
769
770
23.8k
    if (size > 0) {
771
23.3k
        if (ctx->xref_table == NULL) {
772
8.38k
            ctx->xref_table = (xref_table_t *)gs_alloc_bytes(ctx->memory, sizeof(xref_table_t), "read_xref_stream allocate xref table");
773
8.38k
            if (ctx->xref_table == NULL)
774
0
                return_error(gs_error_VMerror);
775
8.38k
            memset(ctx->xref_table, 0x00, sizeof(xref_table_t));
776
777
8.38k
            ctx->xref_table->xref = (xref_entry *)gs_alloc_bytes(ctx->memory, ((size_t)start + (size_t)size) * (size_t)sizeof(xref_entry), "read_xref_stream allocate xref table entries");
778
8.38k
            if (ctx->xref_table->xref == NULL){
779
23
                gs_free_object(ctx->memory, ctx->xref_table, "free xref table on error allocating entries");
780
23
                ctx->xref_table = NULL;
781
23
                return_error(gs_error_VMerror);
782
23
            }
783
#if REFCNT_DEBUG
784
            ctx->xref_table->UID = ctx->ref_UID++;
785
            outprintf(ctx->memory, "Allocated xref table with UID %"PRIi64"\n", ctx->xref_table->UID);
786
#endif
787
788
8.36k
            memset(ctx->xref_table->xref, 0x00, (start + size) * sizeof(xref_entry));
789
8.36k
            ctx->xref_table->ctx = ctx;
790
8.36k
            ctx->xref_table->type = PDF_XREF_TABLE;
791
8.36k
            ctx->xref_table->xref_size = start + size;
792
8.36k
            pdfi_countup(ctx->xref_table);
793
15.0k
        } else {
794
15.0k
            if (start + size > ctx->xref_table->xref_size) {
795
11.0k
                code = resize_xref(ctx, start + size);
796
11.0k
                if (code < 0)
797
16
                    return code;
798
11.0k
            }
799
15.0k
        }
800
23.3k
    }
801
802
23.8k
    pdfi_skip_white(ctx, s);
803
337k
    for (i=0;i< size;i++){
804
314k
        xref_entry *entry = &ctx->xref_table->xref[i + start];
805
314k
        unsigned char free;
806
314k
        gs_offset_t off;
807
314k
        unsigned int gen;
808
809
314k
        bytes = pdfi_read_bytes(ctx, (byte *)Buffer, 1, 20, s);
810
314k
        if (bytes < 20)
811
5
            return_error(gs_error_ioerror);
812
314k
        j = 19;
813
314k
        if ((Buffer[19] != 0x0a && Buffer[19] != 0x0d) || (Buffer[18] != 0x0d && Buffer[18] != 0x0a && Buffer[18] != 0x20))
814
13.7k
            pdfi_set_warning(ctx, 0, NULL, W_PDF_BAD_XREF_ENTRY_SIZE, "read_xref_section", NULL);
815
334k
        while (Buffer[j] != 0x0D && Buffer[j] != 0x0A) {
816
21.3k
            pdfi_unread_byte(ctx, s, (byte)Buffer[j]);
817
21.3k
            if (--j < 0) {
818
649
                pdfi_set_warning(ctx, 0, NULL, W_PDF_BAD_XREF_ENTRY_NO_EOL, "read_xref_section", NULL);
819
649
                outprintf(ctx->memory, "Invalid xref entry, line terminator missing.\n");
820
649
                code = read_xref_entry_slow(ctx, s, &off, &gen, &free);
821
649
                if (code < 0)
822
331
                    return code;
823
318
                code = write_offset((byte *)Buffer, off, gen, free);
824
318
                if (code < 0)
825
0
                    return code;
826
318
                j = 19;
827
318
                break;
828
318
            }
829
21.3k
        }
830
313k
        Buffer[j] = 0x00;
831
313k
        if (entry->object_num != 0)
832
6.54k
            continue;
833
834
307k
        if (sscanf(Buffer, "%"PRIdOFFSET" %d %c", &entry->u.uncompressed.offset, &entry->u.uncompressed.generation_num, &free) != 3) {
835
718
            pdfi_set_warning(ctx, 0, NULL, W_PDF_BAD_XREF_ENTRY_FORMAT, "read_xref_section", NULL);
836
718
            outprintf(ctx->memory, "Invalid xref entry, incorrect format.\n");
837
718
            pdfi_unread(ctx, s, (byte *)Buffer, 20);
838
718
            code = read_xref_entry_slow(ctx, s, &off, &gen, &free);
839
718
            if (code < 0)
840
433
                return code;
841
285
            code = write_offset((byte *)Buffer, off, gen, free);
842
285
            if (code < 0)
843
0
                return code;
844
285
        }
845
846
306k
        entry->compressed = false;
847
306k
        entry->object_num = i + start;
848
306k
        if (free == 'f')
849
89.6k
            entry->free = true;
850
306k
        if(free == 'n')
851
216k
            entry->free = false;
852
306k
        if (entry->object_num == 0) {
853
5.69k
            if (!entry->free) {
854
60
                pdfi_set_warning(ctx, 0, NULL, W_PDF_XREF_OBJECT0_NOT_FREE, "read_xref_section", NULL);
855
60
            }
856
5.69k
        }
857
306k
    }
858
859
23.0k
    return 0;
860
23.8k
}
861
862
static int read_xref(pdf_context *ctx, pdf_c_stream *s)
863
9.51k
{
864
9.51k
    int code = 0;
865
9.51k
    pdf_dict *d = NULL;
866
9.51k
    uint64_t max_obj = 0;
867
9.51k
    int64_t num, XRefStm = 0;
868
9.51k
    int obj_num;
869
9.51k
    bool known = false;
870
871
9.51k
    if (ctx->repaired)
872
2
        return 0;
873
874
32.5k
    do {
875
32.5k
        uint64_t section_start, section_size;
876
877
32.5k
        code = read_xref_section(ctx, s, &section_start, &section_size);
878
32.5k
        if (code < 0)
879
1.01k
            return code;
880
881
31.5k
        if (section_size > 0 && section_start + section_size - 1 > max_obj)
882
20.3k
            max_obj = section_start + section_size - 1;
883
884
        /* code == 1 => read_xref_section ended with a trailer. */
885
31.5k
    } while (code != 1);
886
887
8.49k
    code = pdfi_read_dict(ctx, ctx->main_stream, 0, 0);
888
8.49k
    if (code < 0)
889
164
        return code;
890
891
8.33k
    d = (pdf_dict *)ctx->stack_top[-1];
892
8.33k
    if (pdfi_type_of(d) != PDF_DICT) {
893
14
        pdfi_pop(ctx, 1);
894
14
        return_error(gs_error_typecheck);
895
14
    }
896
8.31k
    pdfi_countup(d);
897
8.31k
    pdfi_pop(ctx, 1);
898
899
    /* We don't want to pollute the Trailer dictionary with any XRefStm key/value pairs
900
     * which will happen when we do pdfi_merge_dicts(). So we get any XRefStm here and
901
     * if there was one, remove it from the dictionary before we merge with the
902
     * primary trailer.
903
     */
904
8.31k
    code = pdfi_dict_get_int(ctx, d, "XRefStm", &XRefStm);
905
8.31k
    if (code < 0 && code != gs_error_undefined)
906
1
        goto error;
907
908
8.31k
    if (code == 0) {
909
270
        code = pdfi_dict_delete(ctx, d, "XRefStm");
910
270
        if (code < 0)
911
0
            goto error;
912
270
    }
913
914
8.31k
    if (ctx->Trailer == NULL) {
915
7.35k
        ctx->Trailer = d;
916
7.35k
        pdfi_countup(d);
917
7.35k
    } else {
918
962
        code = pdfi_merge_dicts(ctx, ctx->Trailer, d);
919
962
        if (code < 0) {
920
0
            if ((code = pdfi_set_error_stop(ctx, code, NULL, E_PDF_BADXREF, "read_xref", "")) < 0) {
921
0
                return code;
922
0
            }
923
0
        }
924
962
    }
925
926
    /* Check if the highest subsection + size exceeds the /Size in the
927
     * trailer dictionary and set a warning flag if it does
928
     */
929
8.31k
    code = pdfi_dict_get_int(ctx, d, "Size", &num);
930
8.31k
    if (code < 0)
931
14
        goto error;
932
933
8.30k
    if (max_obj >= num)
934
509
        pdfi_set_warning(ctx, 0, NULL, W_PDF_BAD_XREF_SIZE, "read_xref", NULL);
935
936
    /* Check if this is a modified file and has any
937
     * previous xref entries.
938
     */
939
8.30k
    code = pdfi_dict_known(ctx, d, "Prev", &known);
940
8.30k
    if (known) {
941
3.57k
        code = pdfi_dict_get_int(ctx, d, "Prev", &num);
942
3.57k
        if (code < 0)
943
12
            goto error;
944
945
3.56k
        if (num < 0 || num > ctx->main_stream_length) {
946
1.23k
            code = gs_note_error(gs_error_rangecheck);
947
1.23k
            goto error;
948
1.23k
        }
949
950
2.32k
        if (pdfi_loop_detector_check_object(ctx, num) == true) {
951
2
            code = gs_note_error(gs_error_circular_reference);
952
2
            goto error;
953
2
        }
954
2.32k
        else {
955
2.32k
            code = pdfi_loop_detector_add_object(ctx, num);
956
2.32k
            if (code < 0)
957
0
                goto error;
958
2.32k
        }
959
960
2.32k
        code = pdfi_seek(ctx, s, num, SEEK_SET);
961
2.32k
        if (code < 0)
962
0
            goto error;
963
964
2.32k
        if (!ctx->repaired) {
965
2.32k
            code = pdfi_read_token(ctx, ctx->main_stream, 0, 0);
966
2.32k
            if (code < 0)
967
94
                goto error;
968
969
2.23k
            if (code == 0) {
970
3
                code = gs_note_error(gs_error_syntaxerror);
971
3
                goto error;
972
3
            }
973
2.23k
        } else {
974
0
            code = 0;
975
0
            goto error;
976
0
        }
977
978
2.22k
        if ((intptr_t)(ctx->stack_top[-1]) == (intptr_t)TOKEN_XREF) {
979
            /* Read old-style xref table */
980
1.02k
            pdfi_pop(ctx, 1);
981
1.02k
            code = read_xref(ctx, ctx->main_stream);
982
1.02k
            if (code < 0)
983
185
                goto error;
984
1.20k
        } else {
985
1.20k
            pdfi_pop(ctx, 1);
986
1.20k
            code = gs_note_error(gs_error_typecheck);
987
1.20k
            goto error;
988
1.20k
        }
989
2.22k
    }
990
991
    /* Now check if this is a hybrid file. */
992
5.56k
    if (XRefStm != 0) {
993
160
        ctx->is_hybrid = true;
994
995
160
        if (ctx->args.pdfdebug)
996
0
            outprintf(ctx->memory, "%% File is a hybrid, containing xref table and xref stream. Reading the stream.\n");
997
998
999
160
        if (pdfi_loop_detector_check_object(ctx, XRefStm) == true) {
1000
0
            code = gs_note_error(gs_error_circular_reference);
1001
0
            goto error;
1002
0
        }
1003
160
        else {
1004
160
            code = pdfi_loop_detector_add_object(ctx, XRefStm);
1005
160
            if (code < 0)
1006
0
                goto error;
1007
160
        }
1008
1009
160
        code = pdfi_loop_detector_mark(ctx);
1010
160
        if (code < 0)
1011
0
            goto error;
1012
1013
        /* Because of the way the code works when we read a file which is a pure
1014
         * xref stream file, we need to read the first integer of 'x y obj'
1015
         * because the xref stream decoding code expects that to be on the stack.
1016
         */
1017
160
        pdfi_seek(ctx, s, XRefStm, SEEK_SET);
1018
1019
160
        code = pdfi_read_bare_int(ctx, ctx->main_stream, &obj_num);
1020
160
        if (code < 0) {
1021
0
            pdfi_set_error(ctx, 0, NULL, E_PDF_BADXREFSTREAM, "read_xref", "");
1022
0
            pdfi_loop_detector_cleartomark(ctx);
1023
0
            goto error;
1024
0
        }
1025
1026
160
        code = pdfi_read_xref_stream_dict(ctx, ctx->main_stream, obj_num);
1027
        /* We could just fall through to the exit here, but choose not to in order to avoid possible mistakes in future */
1028
160
        if (code < 0) {
1029
16
            pdfi_loop_detector_cleartomark(ctx);
1030
16
            goto error;
1031
16
        }
1032
1033
144
        pdfi_loop_detector_cleartomark(ctx);
1034
144
    } else
1035
5.40k
        code = 0;
1036
1037
8.31k
error:
1038
8.31k
    pdfi_countdown(d);
1039
8.31k
    return code;
1040
5.56k
}
1041
1042
int pdfi_read_xref(pdf_context *ctx)
1043
83.8k
{
1044
83.8k
    int code = 0;
1045
83.8k
    int obj_num;
1046
1047
83.8k
    code = pdfi_loop_detector_mark(ctx);
1048
83.8k
    if (code < 0)
1049
0
        return code;
1050
1051
83.8k
    if (ctx->startxref == 0)
1052
48.1k
        goto repair;
1053
1054
35.6k
    code = pdfi_loop_detector_add_object(ctx, ctx->startxref);
1055
35.6k
    if (code < 0)
1056
0
        goto exit;
1057
1058
35.6k
    if (ctx->args.pdfdebug)
1059
0
        outprintf(ctx->memory, "%% Trying to read 'xref' token for xref table, or 'int int obj' for an xref stream\n");
1060
1061
35.6k
    if (ctx->startxref > ctx->main_stream_length - 5) {
1062
9.05k
        if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_rangecheck), NULL, E_PDF_BADSTARTXREF, "pdfi_read_xref", (char *)"startxref offset is beyond end of file")) < 0)
1063
0
            goto exit;
1064
1065
9.05k
        goto repair;
1066
9.05k
    }
1067
26.6k
    if (ctx->startxref < 0) {
1068
323
        if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_rangecheck), NULL, E_PDF_BADSTARTXREF, "pdfi_read_xref", (char *)"startxref offset is before start of file")) < 0)
1069
0
            goto exit;
1070
1071
323
        goto repair;
1072
323
    }
1073
1074
    /* Read the xref(s) */
1075
26.2k
    pdfi_seek(ctx, ctx->main_stream, ctx->startxref, SEEK_SET);
1076
1077
    /* If it starts with an int, it's an xref stream dict */
1078
26.2k
    code = pdfi_read_bare_int(ctx, ctx->main_stream, &obj_num);
1079
26.2k
    if (code == 1) {
1080
10.0k
        if (pdfi_check_xref_stream(ctx)) {
1081
7.54k
            code = pdfi_read_xref_stream_dict(ctx, ctx->main_stream, obj_num);
1082
7.54k
            if (code < 0)
1083
2.84k
                goto repair;
1084
7.54k
        } else
1085
2.53k
            goto repair;
1086
16.2k
    } else {
1087
        /* If not, it had better start 'xref', and be an old-style xref table */
1088
16.2k
        code = pdfi_read_bare_keyword(ctx, ctx->main_stream);
1089
16.2k
        if (code != TOKEN_XREF) {
1090
7.75k
            if ((code = pdfi_set_error_stop(ctx, gs_note_error(gs_error_syntaxerror), NULL, E_PDF_BADSTARTXREF, "pdfi_read_xref", (char *)"Failed to read any token at the startxref location")) < 0)
1091
0
                goto exit;
1092
1093
7.75k
            goto repair;
1094
7.75k
        }
1095
1096
8.45k
        code = read_xref(ctx, ctx->main_stream);
1097
8.45k
        if (code < 0)
1098
3.75k
            goto repair;
1099
8.45k
    }
1100
1101
9.40k
    if(ctx->args.pdfdebug && ctx->xref_table) {
1102
0
        int i, j;
1103
0
        xref_entry *entry;
1104
0
        char Buffer[32];
1105
1106
0
        outprintf(ctx->memory, "\n%% Dumping xref table\n");
1107
0
        for (i=0;i < ctx->xref_table->xref_size;i++) {
1108
0
            entry = &ctx->xref_table->xref[i];
1109
0
            if(entry->compressed) {
1110
0
                outprintf(ctx->memory, "*");
1111
0
                gs_snprintf(Buffer, sizeof(Buffer), "%"PRId64"", entry->object_num);
1112
0
                j = 10 - strlen(Buffer);
1113
0
                while(j--) {
1114
0
                    outprintf(ctx->memory, " ");
1115
0
                }
1116
0
                outprintf(ctx->memory, "%s ", Buffer);
1117
1118
0
                gs_snprintf(Buffer, sizeof(Buffer), "%ld", entry->u.compressed.compressed_stream_num);
1119
0
                j = 10 - strlen(Buffer);
1120
0
                while(j--) {
1121
0
                    outprintf(ctx->memory, " ");
1122
0
                }
1123
0
                outprintf(ctx->memory, "%s ", Buffer);
1124
1125
0
                gs_snprintf(Buffer, sizeof(Buffer), "%ld", entry->u.compressed.object_index);
1126
0
                j = 10 - strlen(Buffer);
1127
0
                while(j--) {
1128
0
                    outprintf(ctx->memory, " ");
1129
0
                }
1130
0
                outprintf(ctx->memory, "%s ", Buffer);
1131
0
            }
1132
0
            else {
1133
0
                outprintf(ctx->memory, " ");
1134
1135
0
                gs_snprintf(Buffer, sizeof(Buffer), "%ld", entry->object_num);
1136
0
                j = 10 - strlen(Buffer);
1137
0
                while(j--) {
1138
0
                    outprintf(ctx->memory, " ");
1139
0
                }
1140
0
                outprintf(ctx->memory, "%s ", Buffer);
1141
1142
0
                gs_snprintf(Buffer, sizeof(Buffer), "%"PRIdOFFSET"", entry->u.uncompressed.offset);
1143
0
                j = 10 - strlen(Buffer);
1144
0
                while(j--) {
1145
0
                    outprintf(ctx->memory, " ");
1146
0
                }
1147
0
                outprintf(ctx->memory, "%s ", Buffer);
1148
1149
0
                gs_snprintf(Buffer, sizeof(Buffer), "%ld", entry->u.uncompressed.generation_num);
1150
0
                j = 10 - strlen(Buffer);
1151
0
                while(j--) {
1152
0
                    outprintf(ctx->memory, " ");
1153
0
                }
1154
0
                outprintf(ctx->memory, "%s ", Buffer);
1155
0
            }
1156
0
            if (entry->free)
1157
0
                outprintf(ctx->memory, "f\n");
1158
0
            else
1159
0
                outprintf(ctx->memory, "n\n");
1160
0
        }
1161
0
    }
1162
9.40k
    if (ctx->args.pdfdebug)
1163
0
        outprintf(ctx->memory, "\n");
1164
1165
9.40k
 exit:
1166
9.40k
    (void)pdfi_loop_detector_cleartomark(ctx);
1167
1168
9.40k
    if (code < 0)
1169
0
        return code;
1170
1171
9.40k
    return 0;
1172
1173
74.4k
repair:
1174
74.4k
    (void)pdfi_loop_detector_cleartomark(ctx);
1175
74.4k
    if (!ctx->repaired && !ctx->args.pdfstoponerror)
1176
74.3k
        return(pdfi_repair_file(ctx));
1177
61
    return 0;
1178
74.4k
}