Coverage Report

Created: 2026-08-31 06:57

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/qpdf/libqpdf/QPDFWriter.cc
Line
Count
Source
1
#include <qpdf/qpdf-config.h> // include early for large file support
2
3
#include <qpdf/QPDFWriter_private.hh>
4
5
#include <qpdf/MD5.hh>
6
#include <qpdf/Pl_AES_PDF.hh>
7
#include <qpdf/Pl_Flate.hh>
8
#include <qpdf/Pl_MD5.hh>
9
#include <qpdf/Pl_PNGFilter.hh>
10
#include <qpdf/Pl_RC4.hh>
11
#include <qpdf/Pl_StdioFile.hh>
12
#include <qpdf/QIntC.hh>
13
#include <qpdf/QPDFObjectHandle_private.hh>
14
#include <qpdf/QPDFObject_private.hh>
15
#include <qpdf/QPDF_private.hh>
16
#include <qpdf/QTC.hh>
17
#include <qpdf/QUtil.hh>
18
#include <qpdf/RC4.hh>
19
#include <qpdf/Util.hh>
20
21
#include <algorithm>
22
#include <concepts>
23
#include <cstdlib>
24
#include <stdexcept>
25
#include <tuple>
26
27
using namespace std::literals;
28
using namespace qpdf;
29
30
using Encryption = impl::Doc::Encryption;
31
using Config = Writer::Config;
32
33
QPDFWriter::ProgressReporter::~ProgressReporter() // NOLINT (modernize-use-equals-default)
34
0
{
35
    // Must be explicit and not inline -- see QPDF_DLL_CLASS in README-maintainer
36
0
}
37
38
QPDFWriter::FunctionProgressReporter::FunctionProgressReporter(std::function<void(int)> handler) :
39
0
    handler(handler)
40
0
{
41
0
}
42
43
QPDFWriter::FunctionProgressReporter::~FunctionProgressReporter() // NOLINT
44
                                                                  // (modernize-use-equals-default)
45
0
{
46
    // Must be explicit and not inline -- see QPDF_DLL_CLASS in README-maintainer
47
0
}
48
49
void
50
QPDFWriter::FunctionProgressReporter::reportProgress(int progress)
51
0
{
52
0
    handler(progress);
53
0
}
54
55
namespace
56
{
57
    class Pl_stack
58
    {
59
        // A pipeline Popper is normally returned by Pl_stack::activate, or, if necessary, a
60
        // reference to a Popper instance can be passed into activate. When the Popper goes out of
61
        // scope, the pipeline stack is popped. This causes finish to be called on the current
62
        // pipeline and the pipeline stack to be popped until the top of stack is a previous active
63
        // top of stack and restores the pipeline to that point. It deletes any pipelines that it
64
        // pops.
65
        class Popper
66
        {
67
            friend class Pl_stack;
68
69
          public:
70
            Popper() = default;
71
            Popper(Popper const&) = delete;
72
            Popper(Popper&& other) noexcept
73
0
            {
74
0
                // For MSVC, default pops the stack
75
0
                if (this != &other) {
76
0
                    stack = other.stack;
77
0
                    stack_id = other.stack_id;
78
0
                    other.stack = nullptr;
79
0
                    other.stack_id = 0;
80
0
                };
81
0
            }
82
            Popper& operator=(Popper const&) = delete;
83
            Popper&
84
            operator=(Popper&& other) noexcept
85
0
            {
86
0
                // For MSVC, default pops the stack
87
0
                if (this != &other) {
88
0
                    stack = other.stack;
89
0
                    stack_id = other.stack_id;
90
0
                    other.stack = nullptr;
91
0
                    other.stack_id = 0;
92
0
                };
93
0
                return *this;
94
0
            }
95
96
            ~Popper();
97
98
            // Manually pop pipeline from the pipeline stack.
99
            void pop();
100
101
          private:
102
            Popper(Pl_stack& stack) :
103
814k
                stack(&stack)
104
814k
            {
105
814k
            }
106
107
            Pl_stack* stack{nullptr};
108
            unsigned long stack_id{0};
109
        };
110
111
      public:
112
        Pl_stack(pl::Count*& top) :
113
76.3k
            top(top)
114
76.3k
        {
115
76.3k
        }
116
117
        Popper
118
        popper()
119
101k
        {
120
101k
            return {*this};
121
101k
        }
122
123
        void
124
        initialize(Pipeline* p)
125
76.3k
        {
126
76.3k
            auto c = std::make_unique<pl::Count>(++last_id, p);
127
76.3k
            top = c.get();
128
76.3k
            stack.emplace_back(std::move(c));
129
76.3k
        }
130
131
        Popper
132
        activate(std::string& str)
133
666k
        {
134
666k
            Popper pp{*this};
135
666k
            activate(pp, str);
136
666k
            return pp;
137
666k
        }
138
139
        void
140
        activate(Popper& pp, std::string& str)
141
666k
        {
142
666k
            activate(pp, false, &str, nullptr);
143
666k
        }
144
145
        void
146
        activate(Popper& pp, std::unique_ptr<Pipeline> next)
147
0
        {
148
0
            count_buffer.clear();
149
0
            activate(pp, false, &count_buffer, std::move(next));
150
0
        }
151
152
        Popper
153
        activate(
154
            bool discard = false,
155
            std::string* str = nullptr,
156
            std::unique_ptr<Pipeline> next = nullptr)
157
46.5k
        {
158
46.5k
            Popper pp{*this};
159
46.5k
            activate(pp, discard, str, std::move(next));
160
46.5k
            return pp;
161
46.5k
        }
162
163
        void
164
        activate(
165
            Popper& pp,
166
            bool discard = false,
167
            std::string* str = nullptr,
168
            std::unique_ptr<Pipeline> next = nullptr)
169
745k
        {
170
745k
            std::unique_ptr<pl::Count> c;
171
745k
            if (next) {
172
0
                c = std::make_unique<pl::Count>(++last_id, count_buffer, std::move(next));
173
745k
            } else if (discard) {
174
78.9k
                c = std::make_unique<pl::Count>(++last_id, nullptr);
175
666k
            } else if (!str) {
176
0
                c = std::make_unique<pl::Count>(++last_id, top);
177
666k
            } else {
178
666k
                c = std::make_unique<pl::Count>(++last_id, *str);
179
666k
            }
180
745k
            pp.stack_id = last_id;
181
745k
            top = c.get();
182
745k
            stack.emplace_back(std::move(c));
183
745k
        }
184
        void
185
        activate_md5(Popper& pp)
186
35.1k
        {
187
35.1k
            qpdf_assert_debug(!md5_pipeline);
188
35.1k
            qpdf_assert_debug(md5_id == 0);
189
35.1k
            qpdf_assert_debug(top->getCount() == 0);
190
35.1k
            md5_pipeline = std::make_unique<Pl_MD5>("qpdf md5", top);
191
35.1k
            md5_pipeline->persistAcrossFinish(true);
192
            // Special case code in pop clears m->md5_pipeline upon deletion.
193
35.1k
            auto c = std::make_unique<pl::Count>(++last_id, md5_pipeline.get());
194
35.1k
            pp.stack_id = last_id;
195
35.1k
            md5_id = last_id;
196
35.1k
            top = c.get();
197
35.1k
            stack.emplace_back(std::move(c));
198
35.1k
        }
199
200
        // Return the hex digest and disable the MD5 pipeline.
201
        std::string
202
        hex_digest()
203
34.0k
        {
204
34.0k
            qpdf_assert_debug(md5_pipeline);
205
34.0k
            auto digest = md5_pipeline->getHexDigest();
206
34.0k
            md5_pipeline->enable(false);
207
34.0k
            return digest;
208
34.0k
        }
209
210
        void
211
        clear_buffer()
212
0
        {
213
0
            count_buffer.clear();
214
0
        }
215
216
      private:
217
        void
218
        pop(unsigned long stack_id)
219
814k
        {
220
814k
            if (!stack_id) {
221
33.5k
                return;
222
33.5k
            }
223
781k
            qpdf_assert_debug(stack.size() >= 2);
224
781k
            top->finish();
225
781k
            qpdf_assert_debug(stack.back().get() == top);
226
            // It used to be possible for this assertion to fail if writeLinearized exits by
227
            // exception when deterministic ID. There are no longer any cases in which two
228
            // dynamically allocated pipeline Popper objects ever exist at the same time, so the
229
            // assertion will fail if they get popped out of order from automatic destruction.
230
781k
            qpdf_assert_debug(top->id() == stack_id);
231
781k
            if (stack_id == md5_id) {
232
35.1k
                md5_pipeline = nullptr;
233
35.1k
                md5_id = 0;
234
35.1k
            }
235
781k
            stack.pop_back();
236
781k
            top = stack.back().get();
237
781k
        }
238
239
        std::vector<std::unique_ptr<pl::Count>> stack;
240
        pl::Count*& top;
241
        std::unique_ptr<Pl_MD5> md5_pipeline{nullptr};
242
        unsigned long last_id{0};
243
        unsigned long md5_id{0};
244
        std::string count_buffer;
245
    };
246
} // namespace
247
248
Pl_stack::Popper::~Popper()
249
814k
{
250
814k
    if (stack) {
251
768k
        stack->pop(stack_id);
252
768k
    }
253
814k
}
254
255
void
256
Pl_stack::Popper::pop()
257
45.8k
{
258
45.8k
    if (stack) {
259
45.8k
        stack->pop(stack_id);
260
45.8k
    }
261
45.8k
    stack_id = 0;
262
45.8k
    stack = nullptr;
263
45.8k
}
264
265
namespace qpdf::impl
266
{
267
    // Writer class is restricted to QPDFWriter so that only it can call certain methods.
268
    class Writer: protected Doc::Common
269
    {
270
      public:
271
        // flags used by unparseObject
272
        static int const f_stream = 1 << 0;
273
        static int const f_filtered = 1 << 1;
274
        static int const f_in_ostream = 1 << 2;
275
        static int const f_hex_string = 1 << 3;
276
        static int const f_no_encryption = 1 << 4;
277
278
        enum trailer_e { t_normal, t_lin_first, t_lin_second };
279
280
        Writer() = delete;
281
        Writer(Writer const&) = delete;
282
        Writer(Writer&&) = delete;
283
        Writer& operator=(Writer const&) = delete;
284
        Writer& operator=(Writer&&) = delete;
285
        ~Writer()
286
76.3k
        {
287
76.3k
            if (file && close_file) {
288
0
                fclose(file);
289
0
            }
290
76.3k
            delete output_buffer;
291
76.3k
        }
292
        Writer(QPDF& qpdf, QPDFWriter& w) :
293
76.3k
            Common(qpdf.doc()),
294
76.3k
            lin(qpdf.doc().linearization()),
295
76.3k
            cfg(true),
296
76.3k
            root_og(qpdf.getRoot().indirect() ? qpdf.getRoot().id_gen() : QPDFObjGen(-1, 0)),
297
76.3k
            pipeline_stack(pipeline)
298
76.3k
        {
299
76.3k
        }
300
301
        void write();
302
        std::map<QPDFObjGen, QPDFXRefEntry> getWrittenXRefTable();
303
        void setMinimumPDFVersion(std::string const& version, int extension_level = 0);
304
        void copyEncryptionParameters(QPDF&);
305
        void doWriteSetup();
306
        void prepareFileForWrite();
307
308
        void disableIncompatibleEncryption(int major, int minor, int extension_level);
309
        void interpretR3EncryptionParameters(
310
            bool allow_accessibility,
311
            bool allow_extract,
312
            bool allow_assemble,
313
            bool allow_annotate_and_form,
314
            bool allow_form_filling,
315
            bool allow_modify_other,
316
            qpdf_r3_print_e print,
317
            qpdf_r3_modify_e modify);
318
        void setEncryptionParameters(char const* user_password, char const* owner_password);
319
        void setEncryptionMinimumVersion();
320
        void parseVersion(std::string const& version, int& major, int& minor) const;
321
        int compareVersions(int major1, int minor1, int major2, int minor2) const;
322
        void generateID(bool encrypted);
323
        std::string getOriginalID1();
324
        void initializeTables(size_t extra = 0);
325
        void preserveObjectStreams();
326
        void generateObjectStreams();
327
        void initializeSpecialStreams();
328
        void enqueue(QPDFObjectHandle const& object);
329
        void enqueueObjectsStandard();
330
        void enqueueObjectsPCLm();
331
        void enqueuePart(std::vector<QPDFObjectHandle>& part);
332
        void assignCompressedObjectNumbers(QPDFObjGen og);
333
        Dictionary trimmed_trailer();
334
335
        // Returns tuple<filter, compress_stream, is_root_metadata>
336
        std::tuple<const bool, const bool, const bool>
337
        will_filter_stream(QPDFObjectHandle stream, std::string* stream_data);
338
339
        // Test whether stream would be filtered if it were written.
340
        bool will_filter_stream(QPDFObjectHandle stream);
341
        unsigned int bytesNeeded(long long n);
342
        void writeBinary(unsigned long long val, unsigned int bytes);
343
        Writer& write(std::string_view str);
344
        Writer& write(size_t count, char c);
345
        Writer& write(std::integral auto val);
346
        Writer& write_name(std::string const& str);
347
        Writer& write_string(std::string const& str, bool force_binary = false);
348
        Writer& write_encrypted(std::string_view str);
349
350
        template <typename... Args>
351
        Writer& write_qdf(Args&&... args);
352
        template <typename... Args>
353
        Writer& write_no_qdf(Args&&... args);
354
        void writeObjectStreamOffsets(std::vector<qpdf_offset_t>& offsets, int first_obj);
355
        void writeObjectStream(QPDFObjectHandle object);
356
        void writeObject(QPDFObjectHandle object, int object_stream_index = -1);
357
        void writeTrailer(
358
            trailer_e which,
359
            int size,
360
            bool xref_stream,
361
            qpdf_offset_t prev,
362
            int linearization_pass);
363
        void unparseObject(
364
            QPDFObjectHandle object,
365
            size_t level,
366
            int flags,
367
            // for stream dictionaries
368
            size_t stream_length = 0,
369
            bool compress = false);
370
        void unparseChild(QPDFObjectHandle const& child, size_t level, int flags);
371
        int openObject(int objid = 0);
372
        void closeObject(int objid);
373
        void writeStandard();
374
        void writeLinearized();
375
        void writeEncryptionDictionary();
376
        void writeHeader();
377
        void writeHintStream(int hint_id);
378
        qpdf_offset_t writeXRefTable(trailer_e which, int first, int last, int size);
379
        qpdf_offset_t writeXRefTable(
380
            trailer_e which,
381
            int first,
382
            int last,
383
            int size,
384
            // for linearization
385
            qpdf_offset_t prev,
386
            bool suppress_offsets,
387
            int hint_id,
388
            qpdf_offset_t hint_offset,
389
            qpdf_offset_t hint_length,
390
            int linearization_pass);
391
        qpdf_offset_t writeXRefStream(
392
            int objid,
393
            int max_id,
394
            qpdf_offset_t max_offset,
395
            trailer_e which,
396
            int first,
397
            int last,
398
            int size);
399
        qpdf_offset_t writeXRefStream(
400
            int objid,
401
            int max_id,
402
            qpdf_offset_t max_offset,
403
            trailer_e which,
404
            int first,
405
            int last,
406
            int size,
407
            // for linearization
408
            qpdf_offset_t prev,
409
            int hint_id,
410
            qpdf_offset_t hint_offset,
411
            qpdf_offset_t hint_length,
412
            bool skip_compression,
413
            int linearization_pass);
414
415
        void setDataKey(int objid);
416
        void indicateProgress(bool decrement, bool finished);
417
        size_t calculateXrefStreamPadding(qpdf_offset_t xref_bytes);
418
419
        void adjustAESStreamLength(size_t& length);
420
        void computeDeterministicIDData();
421
422
      protected:
423
        Doc::Linearization& lin;
424
425
        qpdf::Writer::Config cfg;
426
427
        QPDFObjGen root_og{-1, 0};
428
        char const* filename{"unspecified"};
429
        FILE* file{nullptr};
430
        bool close_file{false};
431
        std::unique_ptr<Pl_Buffer> buffer_pipeline{nullptr};
432
        Buffer* output_buffer{nullptr};
433
434
        std::unique_ptr<QPDF::Doc::Encryption> encryption;
435
        std::string encryption_key;
436
437
        std::string id1; // for /ID key of
438
        std::string id2; // trailer dictionary
439
        std::string final_pdf_version;
440
        int final_extension_level{0};
441
        std::string min_pdf_version;
442
        int min_extension_level{0};
443
        int encryption_dict_objid{0};
444
        std::string cur_data_key;
445
        std::unique_ptr<Pipeline> file_pl;
446
        qpdf::pl::Count* pipeline{nullptr};
447
        std::vector<QPDFObjectHandle> object_queue;
448
        size_t object_queue_front{0};
449
        QPDFWriter::ObjTable obj;
450
        QPDFWriter::NewObjTable new_obj;
451
        int next_objid{1};
452
        int cur_stream_length_id{0};
453
        size_t cur_stream_length{0};
454
        bool added_newline{false};
455
        size_t max_ostream_index{0};
456
        std::set<QPDFObjGen> normalized_streams;
457
        std::map<QPDFObjGen, int> page_object_to_seq;
458
        std::map<QPDFObjGen, int> contents_to_page_seq;
459
        std::map<int, std::vector<QPDFObjGen>> object_stream_to_objects;
460
        Pl_stack pipeline_stack;
461
        std::string deterministic_id_data;
462
        bool did_write_setup{false};
463
464
        // For progress reporting
465
        std::shared_ptr<QPDFWriter::ProgressReporter> progress_reporter;
466
        int events_expected{0};
467
        int events_seen{0};
468
        int next_progress_report{0};
469
    }; // class qpdf::impl::Writer
470
471
} // namespace qpdf::impl
472
473
class QPDFWriter::Members: impl::Writer
474
{
475
    friend class QPDFWriter;
476
    friend class qpdf::Writer;
477
478
  public:
479
    Members(QPDFWriter& w, QPDF& qpdf) :
480
76.3k
        impl::Writer(qpdf, w)
481
76.3k
    {
482
76.3k
    }
483
};
484
485
qpdf::Writer::Writer(QPDF& qpdf, Config cfg) :
486
0
    QPDFWriter(qpdf)
487
0
{
488
0
    m->cfg = cfg;
489
0
}
490
QPDFWriter::QPDFWriter(QPDF& pdf) :
491
76.3k
    m(std::make_shared<Members>(*this, pdf))
492
76.3k
{
493
76.3k
}
494
495
QPDFWriter::QPDFWriter(QPDF& pdf, char const* filename) :
496
0
    m(std::make_shared<Members>(*this, pdf))
497
0
{
498
0
    setOutputFilename(filename);
499
0
}
500
501
QPDFWriter::QPDFWriter(QPDF& pdf, char const* description, FILE* file, bool close_file) :
502
0
    m(std::make_shared<Members>(*this, pdf))
503
0
{
504
0
    setOutputFile(description, file, close_file);
505
0
}
506
507
void
508
QPDFWriter::setOutputFilename(char const* filename)
509
0
{
510
0
    char const* description = filename;
511
0
    FILE* f = nullptr;
512
0
    bool close_file = false;
513
0
    if (filename == nullptr) {
514
0
        description = "standard output";
515
0
        f = stdout;
516
0
        QUtil::binary_stdout();
517
0
    } else {
518
0
        f = QUtil::safe_fopen(filename, "wb+");
519
0
        close_file = true;
520
0
    }
521
0
    setOutputFile(description, f, close_file);
522
0
}
523
524
void
525
QPDFWriter::setOutputFile(char const* description, FILE* file, bool close_file)
526
0
{
527
0
    m->filename = description;
528
0
    m->file = file;
529
0
    m->close_file = close_file;
530
0
    m->file_pl = std::make_unique<Pl_StdioFile>("qpdf output", file);
531
0
    m->pipeline_stack.initialize(m->file_pl.get());
532
0
}
533
534
void
535
QPDFWriter::setOutputMemory()
536
0
{
537
0
    m->filename = "memory buffer";
538
0
    m->buffer_pipeline = std::make_unique<Pl_Buffer>("qpdf output");
539
0
    m->pipeline_stack.initialize(m->buffer_pipeline.get());
540
0
}
541
542
Buffer*
543
QPDFWriter::getBuffer()
544
0
{
545
0
    Buffer* result = m->output_buffer;
546
0
    m->output_buffer = nullptr;
547
0
    return result;
548
0
}
549
550
std::shared_ptr<Buffer>
551
QPDFWriter::getBufferSharedPointer()
552
0
{
553
0
    return std::shared_ptr<Buffer>(getBuffer());
554
0
}
555
556
void
557
QPDFWriter::setOutputPipeline(Pipeline* p)
558
76.3k
{
559
76.3k
    m->filename = "custom pipeline";
560
76.3k
    m->pipeline_stack.initialize(p);
561
76.3k
}
562
563
void
564
QPDFWriter::setObjectStreamMode(qpdf_object_stream_e mode)
565
37.6k
{
566
37.6k
    m->cfg.object_streams(mode);
567
37.6k
}
568
569
void
570
QPDFWriter::setStreamDataMode(qpdf_stream_data_e mode)
571
0
{
572
0
    m->cfg.stream_data(mode);
573
0
}
574
575
Config&
576
Config::stream_data(qpdf_stream_data_e mode)
577
0
{
578
0
    switch (mode) {
579
0
    case qpdf_s_uncompress:
580
0
        decode_level(std::max(qpdf_dl_generalized, decode_level_));
581
0
        compress_streams(false);
582
0
        return *this;
583
584
0
    case qpdf_s_preserve:
585
0
        decode_level(qpdf_dl_none);
586
0
        compress_streams(false);
587
0
        return *this;
588
589
0
    case qpdf_s_compress:
590
0
        decode_level(std::max(qpdf_dl_generalized, decode_level_));
591
0
        compress_streams(true);
592
0
    }
593
0
    return *this;
594
0
}
595
596
void
597
QPDFWriter::setCompressStreams(bool val)
598
0
{
599
0
    m->cfg.compress_streams(val);
600
0
}
601
602
Config&
603
Config::compress_streams(bool val)
604
19.4k
{
605
19.4k
    if (pclm_) {
606
0
        usage("compress_streams cannot be set when pclm is set");
607
0
        return *this;
608
0
    }
609
19.4k
    compress_streams_set_ = true;
610
19.4k
    compress_streams_ = val;
611
19.4k
    return *this;
612
19.4k
}
613
614
void
615
QPDFWriter::setDecodeLevel(qpdf_stream_decode_level_e val)
616
76.3k
{
617
76.3k
    m->cfg.decode_level(val);
618
76.3k
}
619
620
Config&
621
Config::decode_level(qpdf_stream_decode_level_e val)
622
76.3k
{
623
76.3k
    if (pclm_) {
624
0
        usage("stream_decode_level cannot be set when pclm is set");
625
0
        return *this;
626
0
    }
627
76.3k
    decode_level_set_ = true;
628
76.3k
    decode_level_ = val;
629
76.3k
    return *this;
630
76.3k
}
631
632
void
633
QPDFWriter::setRecompressFlate(bool val)
634
0
{
635
0
    m->cfg.recompress_flate(val);
636
0
}
637
638
void
639
QPDFWriter::setContentNormalization(bool val)
640
0
{
641
0
    m->cfg.normalize_content(val);
642
0
}
643
644
void
645
QPDFWriter::setQDFMode(bool val)
646
19.4k
{
647
19.4k
    m->cfg.qdf(val);
648
19.4k
}
649
650
Config&
651
Config::qdf(bool val)
652
59.0k
{
653
59.0k
    if (pclm_ || linearize_) {
654
39.6k
        usage("qdf cannot be set when linearize or pclm are set");
655
39.6k
    }
656
59.0k
    if (preserve_encryption_) {
657
59.0k
        usage("preserve_encryption cannot be set when qdf is set");
658
59.0k
    }
659
59.0k
    qdf_ = val;
660
59.0k
    if (val) {
661
19.4k
        if (!normalize_content_set_) {
662
19.4k
            normalize_content(true);
663
19.4k
        }
664
19.4k
        if (!compress_streams_set_) {
665
19.4k
            compress_streams(false);
666
19.4k
        }
667
19.4k
        if (!decode_level_set_) {
668
0
            decode_level(qpdf_dl_generalized);
669
0
        }
670
19.4k
        preserve_encryption_ = false;
671
        // Generate indirect stream lengths for qdf mode since fix-qdf uses them for storing
672
        // recomputed stream length data. Certain streams such as object streams, xref streams, and
673
        // hint streams always get direct stream lengths.
674
19.4k
        direct_stream_lengths_ = false;
675
19.4k
    }
676
59.0k
    return *this;
677
59.0k
}
678
679
void
680
QPDFWriter::setPreserveUnreferencedObjects(bool val)
681
0
{
682
0
    m->cfg.preserve_unreferenced(val);
683
0
}
684
685
void
686
QPDFWriter::setNewlineBeforeEndstream(bool val)
687
0
{
688
0
    m->cfg.newline_before_endstream(val);
689
0
}
690
691
void
692
QPDFWriter::setMinimumPDFVersion(std::string const& version, int extension_level)
693
0
{
694
0
    m->setMinimumPDFVersion(version, extension_level);
695
0
}
696
697
void
698
impl::Writer::setMinimumPDFVersion(std::string const& version, int extension_level)
699
132k
{
700
132k
    bool set_version = false;
701
132k
    bool set_extension_level = false;
702
132k
    if (min_pdf_version.empty()) {
703
75.9k
        set_version = true;
704
75.9k
        set_extension_level = true;
705
75.9k
    } else {
706
56.7k
        int old_major = 0;
707
56.7k
        int old_minor = 0;
708
56.7k
        int min_major = 0;
709
56.7k
        int min_minor = 0;
710
56.7k
        parseVersion(version, old_major, old_minor);
711
56.7k
        parseVersion(min_pdf_version, min_major, min_minor);
712
56.7k
        int compare = compareVersions(old_major, old_minor, min_major, min_minor);
713
56.7k
        if (compare > 0) {
714
2.88k
            QTC::TC("qpdf", "QPDFWriter increasing minimum version", extension_level == 0 ? 0 : 1);
715
2.88k
            set_version = true;
716
2.88k
            set_extension_level = true;
717
53.8k
        } else if (compare == 0) {
718
2.03k
            if (extension_level > min_extension_level) {
719
72
                set_extension_level = true;
720
72
            }
721
2.03k
        }
722
56.7k
    }
723
724
132k
    if (set_version) {
725
78.8k
        min_pdf_version = version;
726
78.8k
    }
727
132k
    if (set_extension_level) {
728
78.9k
        min_extension_level = extension_level;
729
78.9k
    }
730
132k
}
731
732
void
733
QPDFWriter::setMinimumPDFVersion(PDFVersion const& v)
734
0
{
735
0
    std::string version;
736
0
    int extension_level;
737
0
    v.getVersion(version, extension_level);
738
0
    setMinimumPDFVersion(version, extension_level);
739
0
}
740
741
void
742
QPDFWriter::forcePDFVersion(std::string const& version, int extension_level)
743
0
{
744
0
    m->cfg.forced_pdf_version(version, extension_level);
745
0
}
746
747
void
748
QPDFWriter::setExtraHeaderText(std::string const& text)
749
0
{
750
0
    m->cfg.extra_header_text(text);
751
0
}
752
753
Config&
754
Config::extra_header_text(std::string const& val)
755
0
{
756
0
    extra_header_text_ = val;
757
0
    if (!extra_header_text_.empty() && extra_header_text_.back() != '\n') {
758
0
        extra_header_text_ += "\n";
759
0
    } else {
760
0
        QTC::TC("qpdf", "QPDFWriter extra header text no newline");
761
0
    }
762
0
    return *this;
763
0
}
764
765
void
766
QPDFWriter::setStaticID(bool val)
767
36.5k
{
768
36.5k
    m->cfg.static_id(val);
769
36.5k
}
770
771
void
772
QPDFWriter::setDeterministicID(bool val)
773
39.8k
{
774
39.8k
    m->cfg.deterministic_id(val);
775
39.8k
}
776
777
void
778
QPDFWriter::setStaticAesIV(bool val)
779
0
{
780
0
    if (val) {
781
0
        Pl_AES_PDF::useStaticIV();
782
0
    }
783
0
}
784
785
void
786
QPDFWriter::setSuppressOriginalObjectIDs(bool val)
787
0
{
788
0
    m->cfg.no_original_object_ids(val);
789
0
}
790
791
void
792
QPDFWriter::setPreserveEncryption(bool val)
793
0
{
794
0
    m->cfg.preserve_encryption(val);
795
0
}
796
797
void
798
QPDFWriter::setLinearization(bool val)
799
39.6k
{
800
39.6k
    m->cfg.linearize(val);
801
39.6k
}
802
803
Config&
804
Config::linearize(bool val)
805
39.6k
{
806
39.6k
    if (pclm_ || qdf_) {
807
0
        usage("linearize cannot be set when qdf or pclm are set");
808
0
        return *this;
809
0
    }
810
39.6k
    linearize_ = val;
811
39.6k
    return *this;
812
39.6k
}
813
814
void
815
QPDFWriter::setLinearizationPass1Filename(std::string const& filename)
816
0
{
817
0
    m->cfg.linearize_pass1(filename);
818
0
}
819
820
void
821
QPDFWriter::setPCLm(bool val)
822
0
{
823
0
    m->cfg.pclm(val);
824
0
}
825
826
Config&
827
Config::pclm(bool val)
828
0
{
829
0
    if (decode_level_set_ || compress_streams_set_ || linearize_) {
830
0
        usage(
831
0
            "pclm cannot be set when stream_decode_level, compress_streams, linearize or qdf are "
832
0
            "set");
833
0
        return *this;
834
0
    }
835
0
    pclm_ = val;
836
0
    if (val) {
837
0
        decode_level_ = qpdf_dl_none;
838
0
        compress_streams_ = false;
839
0
        linearize_ = false;
840
0
    }
841
842
0
    return *this;
843
0
}
844
845
void
846
QPDFWriter::setR2EncryptionParametersInsecure(
847
    char const* user_password,
848
    char const* owner_password,
849
    bool allow_print,
850
    bool allow_modify,
851
    bool allow_extract,
852
    bool allow_annotate)
853
0
{
854
0
    m->encryption = std::make_unique<Encryption>(1, 2, 5, true);
855
0
    if (!allow_print) {
856
0
        m->encryption->setP(3, false);
857
0
    }
858
0
    if (!allow_modify) {
859
0
        m->encryption->setP(4, false);
860
0
    }
861
0
    if (!allow_extract) {
862
0
        m->encryption->setP(5, false);
863
0
    }
864
0
    if (!allow_annotate) {
865
0
        m->encryption->setP(6, false);
866
0
    }
867
0
    m->setEncryptionParameters(user_password, owner_password);
868
0
}
869
870
void
871
QPDFWriter::setR3EncryptionParametersInsecure(
872
    char const* user_password,
873
    char const* owner_password,
874
    bool allow_accessibility,
875
    bool allow_extract,
876
    bool allow_assemble,
877
    bool allow_annotate_and_form,
878
    bool allow_form_filling,
879
    bool allow_modify_other,
880
    qpdf_r3_print_e print)
881
17.2k
{
882
17.2k
    m->encryption = std::make_unique<Encryption>(2, 3, 16, true);
883
17.2k
    m->interpretR3EncryptionParameters(
884
17.2k
        allow_accessibility,
885
17.2k
        allow_extract,
886
17.2k
        allow_assemble,
887
17.2k
        allow_annotate_and_form,
888
17.2k
        allow_form_filling,
889
17.2k
        allow_modify_other,
890
17.2k
        print,
891
17.2k
        qpdf_r3m_all);
892
17.2k
    m->setEncryptionParameters(user_password, owner_password);
893
17.2k
}
894
895
void
896
QPDFWriter::setR4EncryptionParametersInsecure(
897
    char const* user_password,
898
    char const* owner_password,
899
    bool allow_accessibility,
900
    bool allow_extract,
901
    bool allow_assemble,
902
    bool allow_annotate_and_form,
903
    bool allow_form_filling,
904
    bool allow_modify_other,
905
    qpdf_r3_print_e print,
906
    bool encrypt_metadata,
907
    bool use_aes)
908
0
{
909
0
    m->encryption = std::make_unique<Encryption>(4, 4, 16, encrypt_metadata);
910
0
    m->cfg.encrypt_use_aes(use_aes);
911
0
    m->interpretR3EncryptionParameters(
912
0
        allow_accessibility,
913
0
        allow_extract,
914
0
        allow_assemble,
915
0
        allow_annotate_and_form,
916
0
        allow_form_filling,
917
0
        allow_modify_other,
918
0
        print,
919
0
        qpdf_r3m_all);
920
0
    m->setEncryptionParameters(user_password, owner_password);
921
0
}
922
923
void
924
QPDFWriter::setR5EncryptionParameters(
925
    char const* user_password,
926
    char const* owner_password,
927
    bool allow_accessibility,
928
    bool allow_extract,
929
    bool allow_assemble,
930
    bool allow_annotate_and_form,
931
    bool allow_form_filling,
932
    bool allow_modify_other,
933
    qpdf_r3_print_e print,
934
    bool encrypt_metadata)
935
0
{
936
0
    m->encryption = std::make_unique<Encryption>(5, 5, 32, encrypt_metadata);
937
0
    m->cfg.encrypt_use_aes(true);
938
0
    m->interpretR3EncryptionParameters(
939
0
        allow_accessibility,
940
0
        allow_extract,
941
0
        allow_assemble,
942
0
        allow_annotate_and_form,
943
0
        allow_form_filling,
944
0
        allow_modify_other,
945
0
        print,
946
0
        qpdf_r3m_all);
947
0
    m->setEncryptionParameters(user_password, owner_password);
948
0
}
949
950
void
951
QPDFWriter::setR6EncryptionParameters(
952
    char const* user_password,
953
    char const* owner_password,
954
    bool allow_accessibility,
955
    bool allow_extract,
956
    bool allow_assemble,
957
    bool allow_annotate_and_form,
958
    bool allow_form_filling,
959
    bool allow_modify_other,
960
    qpdf_r3_print_e print,
961
    bool encrypt_metadata)
962
19.2k
{
963
19.2k
    m->encryption = std::make_unique<Encryption>(5, 6, 32, encrypt_metadata);
964
19.2k
    m->interpretR3EncryptionParameters(
965
19.2k
        allow_accessibility,
966
19.2k
        allow_extract,
967
19.2k
        allow_assemble,
968
19.2k
        allow_annotate_and_form,
969
19.2k
        allow_form_filling,
970
19.2k
        allow_modify_other,
971
19.2k
        print,
972
19.2k
        qpdf_r3m_all);
973
19.2k
    m->cfg.encrypt_use_aes(true);
974
19.2k
    m->setEncryptionParameters(user_password, owner_password);
975
19.2k
}
976
977
void
978
impl::Writer::interpretR3EncryptionParameters(
979
    bool allow_accessibility,
980
    bool allow_extract,
981
    bool allow_assemble,
982
    bool allow_annotate_and_form,
983
    bool allow_form_filling,
984
    bool allow_modify_other,
985
    qpdf_r3_print_e print,
986
    qpdf_r3_modify_e modify)
987
36.5k
{
988
    // Acrobat 5 security options:
989
990
    // Checkboxes:
991
    //   Enable Content Access for the Visually Impaired
992
    //   Allow Content Copying and Extraction
993
994
    // Allowed changes menu:
995
    //   None
996
    //   Only Document Assembly
997
    //   Only Form Field Fill-in or Signing
998
    //   Comment Authoring, Form Field Fill-in or Signing
999
    //   General Editing, Comment and Form Field Authoring
1000
1001
    // Allowed printing menu:
1002
    //   None
1003
    //   Low Resolution
1004
    //   Full printing
1005
1006
    // Meanings of bits in P when R >= 3
1007
    //
1008
    //  3: low-resolution printing
1009
    //  4: document modification except as controlled by 6, 9, and 11
1010
    //  5: extraction
1011
    //  6: add/modify annotations (comment), fill in forms
1012
    //     if 4+6 are set, also allows modification of form fields
1013
    //  9: fill in forms even if 6 is clear
1014
    // 10: accessibility; ignored by readers, should always be set
1015
    // 11: document assembly even if 4 is clear
1016
    // 12: high-resolution printing
1017
36.5k
    if (!allow_accessibility && encryption->getR() <= 3) {
1018
        // Bit 10 is deprecated and should always be set.  This used to mean accessibility.  There
1019
        // is no way to disable accessibility with R > 3.
1020
0
        encryption->setP(10, false);
1021
0
    }
1022
36.5k
    if (!allow_extract) {
1023
0
        encryption->setP(5, false);
1024
0
    }
1025
1026
36.5k
    switch (print) {
1027
0
    case qpdf_r3p_none:
1028
0
        encryption->setP(3, false); // any printing
1029
0
        [[fallthrough]];
1030
0
    case qpdf_r3p_low:
1031
0
        encryption->setP(12, false); // high resolution printing
1032
0
        [[fallthrough]];
1033
36.5k
    case qpdf_r3p_full:
1034
36.5k
        break;
1035
        // no default so gcc warns for missing cases
1036
36.5k
    }
1037
1038
    // Modify options. The qpdf_r3_modify_e options control groups of bits and lack the full
1039
    // flexibility of the spec. This is unfortunate, but it's been in the API for ages, and we're
1040
    // stuck with it. See also allow checks below to control the bits individually.
1041
1042
    // NOT EXERCISED IN TEST SUITE
1043
36.5k
    switch (modify) {
1044
0
    case qpdf_r3m_none:
1045
0
        encryption->setP(11, false); // document assembly
1046
0
        [[fallthrough]];
1047
0
    case qpdf_r3m_assembly:
1048
0
        encryption->setP(9, false); // filling in form fields
1049
0
        [[fallthrough]];
1050
0
    case qpdf_r3m_form:
1051
0
        encryption->setP(6, false); // modify annotations, fill in form fields
1052
0
        [[fallthrough]];
1053
0
    case qpdf_r3m_annotate:
1054
0
        encryption->setP(4, false); // other modifications
1055
0
        [[fallthrough]];
1056
36.5k
    case qpdf_r3m_all:
1057
36.5k
        break;
1058
        // no default so gcc warns for missing cases
1059
36.5k
    }
1060
    // END NOT EXERCISED IN TEST SUITE
1061
1062
36.5k
    if (!allow_assemble) {
1063
0
        encryption->setP(11, false);
1064
0
    }
1065
36.5k
    if (!allow_annotate_and_form) {
1066
0
        encryption->setP(6, false);
1067
0
    }
1068
36.5k
    if (!allow_form_filling) {
1069
0
        encryption->setP(9, false);
1070
0
    }
1071
36.5k
    if (!allow_modify_other) {
1072
0
        encryption->setP(4, false);
1073
0
    }
1074
36.5k
}
1075
1076
void
1077
impl::Writer::setEncryptionParameters(char const* user_password, char const* owner_password)
1078
36.5k
{
1079
36.5k
    generateID(true);
1080
36.5k
    encryption->setId1(id1);
1081
36.5k
    encryption_key = encryption->compute_parameters(user_password, owner_password);
1082
36.5k
    setEncryptionMinimumVersion();
1083
36.5k
}
1084
1085
void
1086
QPDFWriter::copyEncryptionParameters(QPDF& qpdf)
1087
0
{
1088
0
    m->copyEncryptionParameters(qpdf);
1089
0
}
1090
1091
void
1092
impl::Writer::copyEncryptionParameters(QPDF& qpdf)
1093
20.3k
{
1094
20.3k
    cfg.preserve_encryption(false);
1095
20.3k
    QPDFObjectHandle trailer = qpdf.getTrailer();
1096
20.3k
    if (trailer.hasKey("/Encrypt")) {
1097
172
        generateID(true);
1098
172
        id1 = trailer.getKey("/ID").getArrayItem(0).getStringValue();
1099
172
        QPDFObjectHandle encrypt = trailer.getKey("/Encrypt");
1100
172
        int V = encrypt.getKey("/V").getIntValueAsInt();
1101
172
        int key_len = 5;
1102
172
        if (V > 1) {
1103
0
            key_len = encrypt.getKey("/Length").getIntValueAsInt() / 8;
1104
0
        }
1105
172
        const bool encrypt_metadata =
1106
172
            encrypt.hasKey("/EncryptMetadata") && encrypt.getKey("/EncryptMetadata").isBool()
1107
172
            ? encrypt.getKey("/EncryptMetadata").getBoolValue()
1108
172
            : true;
1109
172
        if (V >= 4) {
1110
            // When copying encryption parameters, use AES even if the original file did not.
1111
            // Acrobat doesn't create files with V >= 4 that don't use AES, and the logic of
1112
            // figuring out whether AES is used or not is complicated with /StmF, /StrF, and /EFF
1113
            // all potentially having different values.
1114
0
            cfg.encrypt_use_aes(true);
1115
0
        }
1116
172
        QTC::TC("qpdf", "QPDFWriter copy encrypt metadata", encrypt_metadata ? 0 : 1);
1117
172
        QTC::TC("qpdf", "QPDFWriter copy use_aes", cfg.encrypt_use_aes() ? 0 : 1);
1118
1119
172
        encryption = std::make_unique<Encryption>(
1120
172
            V,
1121
172
            encrypt.getKey("/R").getIntValueAsInt(),
1122
172
            key_len,
1123
172
            static_cast<int>(encrypt.getKey("/P").getIntValue()),
1124
172
            encrypt.getKey("/O").getStringValue(),
1125
172
            encrypt.getKey("/U").getStringValue(),
1126
172
            V < 5 ? "" : encrypt.getKey("/OE").getStringValue(),
1127
172
            V < 5 ? "" : encrypt.getKey("/UE").getStringValue(),
1128
172
            V < 5 ? "" : encrypt.getKey("/Perms").getStringValue(),
1129
172
            id1, // id1 == the other file's id1
1130
172
            encrypt_metadata);
1131
172
        encryption_key = V >= 5 ? qpdf.getEncryptionKey()
1132
172
                                : encryption->compute_encryption_key(qpdf.getPaddedUserPassword());
1133
172
        setEncryptionMinimumVersion();
1134
172
    }
1135
20.3k
}
1136
1137
void
1138
impl::Writer::disableIncompatibleEncryption(int major, int minor, int extension_level)
1139
0
{
1140
0
    if (!encryption) {
1141
0
        return;
1142
0
    }
1143
0
    if (compareVersions(major, minor, 1, 3) < 0) {
1144
0
        encryption = nullptr;
1145
0
        return;
1146
0
    }
1147
0
    int V = encryption->getV();
1148
0
    int R = encryption->getR();
1149
0
    if (compareVersions(major, minor, 1, 4) < 0) {
1150
0
        if (V > 1 || R > 2) {
1151
0
            encryption = nullptr;
1152
0
        }
1153
0
    } else if (compareVersions(major, minor, 1, 5) < 0) {
1154
0
        if (V > 2 || R > 3) {
1155
0
            encryption = nullptr;
1156
0
        }
1157
0
    } else if (compareVersions(major, minor, 1, 6) < 0) {
1158
0
        if (cfg.encrypt_use_aes()) {
1159
0
            encryption = nullptr;
1160
0
        }
1161
0
    } else if (
1162
0
        (compareVersions(major, minor, 1, 7) < 0) ||
1163
0
        ((compareVersions(major, minor, 1, 7) == 0) && extension_level < 3)) {
1164
0
        if (V >= 5 || R >= 5) {
1165
0
            encryption = nullptr;
1166
0
        }
1167
0
    }
1168
1169
0
    if (!encryption) {
1170
0
        QTC::TC("qpdf", "QPDFWriter forced version disabled encryption");
1171
0
    }
1172
0
}
1173
1174
void
1175
impl::Writer::parseVersion(std::string const& version, int& major, int& minor) const
1176
113k
{
1177
113k
    major = QUtil::string_to_int(version.c_str());
1178
113k
    minor = 0;
1179
113k
    size_t p = version.find('.');
1180
113k
    if ((p != std::string::npos) && (version.length() > p)) {
1181
113k
        minor = QUtil::string_to_int(version.substr(p + 1).c_str());
1182
113k
    }
1183
113k
    std::string tmp = std::to_string(major) + "." + std::to_string(minor);
1184
113k
    if (tmp != version) {
1185
        // The version number in the input is probably invalid. This happens with some files that
1186
        // are designed to exercise bugs, such as files in the fuzzer corpus. Unfortunately
1187
        // QPDFWriter doesn't have a way to give a warning, so we just ignore this case.
1188
661
    }
1189
113k
}
1190
1191
int
1192
impl::Writer::compareVersions(int major1, int minor1, int major2, int minor2) const
1193
56.6k
{
1194
56.6k
    if (major1 < major2) {
1195
538
        return -1;
1196
538
    }
1197
56.0k
    if (major1 > major2) {
1198
540
        return 1;
1199
540
    }
1200
55.5k
    if (minor1 < minor2) {
1201
51.1k
        return -1;
1202
51.1k
    }
1203
4.37k
    return minor1 > minor2 ? 1 : 0;
1204
55.5k
}
1205
1206
void
1207
impl::Writer::setEncryptionMinimumVersion()
1208
36.5k
{
1209
36.5k
    auto const R = encryption->getR();
1210
36.5k
    if (R >= 6) {
1211
19.2k
        setMinimumPDFVersion("1.7", 8);
1212
19.2k
    } else if (R == 5) {
1213
0
        setMinimumPDFVersion("1.7", 3);
1214
17.2k
    } else if (R == 4) {
1215
0
        setMinimumPDFVersion(cfg.encrypt_use_aes() ? "1.6" : "1.5");
1216
17.2k
    } else if (R == 3) {
1217
17.2k
        setMinimumPDFVersion("1.4");
1218
17.2k
    } else {
1219
0
        setMinimumPDFVersion("1.3");
1220
0
    }
1221
36.5k
}
1222
1223
void
1224
impl::Writer::setDataKey(int objid)
1225
1.06M
{
1226
1.06M
    if (encryption) {
1227
625k
        cur_data_key = QPDF::compute_data_key(
1228
625k
            encryption_key,
1229
625k
            objid,
1230
625k
            0,
1231
625k
            cfg.encrypt_use_aes(),
1232
625k
            encryption->getV(),
1233
625k
            encryption->getR());
1234
625k
    }
1235
1.06M
}
1236
1237
unsigned int
1238
impl::Writer::bytesNeeded(long long n)
1239
181k
{
1240
181k
    unsigned int bytes = 0;
1241
414k
    while (n) {
1242
233k
        ++bytes;
1243
233k
        n >>= 8;
1244
233k
    }
1245
181k
    return bytes;
1246
181k
}
1247
1248
void
1249
impl::Writer::writeBinary(unsigned long long val, unsigned int bytes)
1250
2.94M
{
1251
2.94M
    if (bytes > sizeof(unsigned long long)) {
1252
0
        throw std::logic_error("QPDFWriter::writeBinary called with too many bytes");
1253
0
    }
1254
2.94M
    unsigned char data[sizeof(unsigned long long)];
1255
7.17M
    for (unsigned int i = 0; i < bytes; ++i) {
1256
4.23M
        data[bytes - i - 1] = static_cast<unsigned char>(val & 0xff);
1257
4.23M
        val >>= 8;
1258
4.23M
    }
1259
2.94M
    pipeline->write(data, bytes);
1260
2.94M
}
1261
1262
impl::Writer&
1263
impl::Writer::write(std::string_view str)
1264
64.9M
{
1265
64.9M
    pipeline->write(str);
1266
64.9M
    return *this;
1267
64.9M
}
1268
1269
impl::Writer&
1270
impl::Writer::write(std::integral auto val)
1271
6.16M
{
1272
6.16M
    pipeline->write(std::to_string(val));
1273
6.16M
    return *this;
1274
6.16M
}
_ZN4qpdf4impl6Writer5writeITkNSt3__18integralEiEERS1_T_
Line
Count
Source
1271
4.21M
{
1272
4.21M
    pipeline->write(std::to_string(val));
1273
4.21M
    return *this;
1274
4.21M
}
_ZN4qpdf4impl6Writer5writeITkNSt3__18integralExEERS1_T_
Line
Count
Source
1271
1.26M
{
1272
1.26M
    pipeline->write(std::to_string(val));
1273
1.26M
    return *this;
1274
1.26M
}
_ZN4qpdf4impl6Writer5writeITkNSt3__18integralEmEERS1_T_
Line
Count
Source
1271
500k
{
1272
500k
    pipeline->write(std::to_string(val));
1273
500k
    return *this;
1274
500k
}
_ZN4qpdf4impl6Writer5writeITkNSt3__18integralEjEERS1_T_
Line
Count
Source
1271
180k
{
1272
180k
    pipeline->write(std::to_string(val));
1273
180k
    return *this;
1274
180k
}
1275
1276
impl::Writer&
1277
impl::Writer::write(size_t count, char c)
1278
138k
{
1279
138k
    pipeline->write(count, c);
1280
138k
    return *this;
1281
138k
}
1282
1283
impl::Writer&
1284
impl::Writer::write_name(std::string const& str)
1285
4.13M
{
1286
4.13M
    pipeline->write(Name::normalize(str));
1287
4.13M
    return *this;
1288
4.13M
}
1289
1290
impl::Writer&
1291
impl::Writer::write_string(std::string const& str, bool force_binary)
1292
386k
{
1293
386k
    pipeline->write(QPDF_String(str).unparse(force_binary));
1294
386k
    return *this;
1295
386k
}
1296
1297
template <typename... Args>
1298
impl::Writer&
1299
impl::Writer::write_qdf(Args&&... args)
1300
3.59M
{
1301
3.59M
    if (cfg.qdf()) {
1302
469k
        pipeline->write(std::forward<Args>(args)...);
1303
469k
    }
1304
3.59M
    return *this;
1305
3.59M
}
qpdf::impl::Writer& qpdf::impl::Writer::write_qdf<char const (&) [2]>(char const (&) [2])
Line
Count
Source
1300
2.68M
{
1301
2.68M
    if (cfg.qdf()) {
1302
375k
        pipeline->write(std::forward<Args>(args)...);
1303
375k
    }
1304
2.68M
    return *this;
1305
2.68M
}
qpdf::impl::Writer& qpdf::impl::Writer::write_qdf<char const (&) [3]>(char const (&) [3])
Line
Count
Source
1300
652k
{
1301
652k
    if (cfg.qdf()) {
1302
55.8k
        pipeline->write(std::forward<Args>(args)...);
1303
55.8k
    }
1304
652k
    return *this;
1305
652k
}
qpdf::impl::Writer& qpdf::impl::Writer::write_qdf<char const (&) [4]>(char const (&) [4])
Line
Count
Source
1300
160k
{
1301
160k
    if (cfg.qdf()) {
1302
18.8k
        pipeline->write(std::forward<Args>(args)...);
1303
18.8k
    }
1304
160k
    return *this;
1305
160k
}
qpdf::impl::Writer& qpdf::impl::Writer::write_qdf<char const (&) [11]>(char const (&) [11])
Line
Count
Source
1300
99.4k
{
1301
99.4k
    if (cfg.qdf()) {
1302
19.2k
        pipeline->write(std::forward<Args>(args)...);
1303
19.2k
    }
1304
99.4k
    return *this;
1305
99.4k
}
1306
1307
template <typename... Args>
1308
impl::Writer&
1309
impl::Writer::write_no_qdf(Args&&... args)
1310
1.23M
{
1311
1.23M
    if (!cfg.qdf()) {
1312
1.10M
        pipeline->write(std::forward<Args>(args)...);
1313
1.10M
    }
1314
1.23M
    return *this;
1315
1.23M
}
qpdf::impl::Writer& qpdf::impl::Writer::write_no_qdf<char const (&) [2]>(char const (&) [2])
Line
Count
Source
1310
1.07M
{
1311
1.07M
    if (!cfg.qdf()) {
1312
959k
        pipeline->write(std::forward<Args>(args)...);
1313
959k
    }
1314
1.07M
    return *this;
1315
1.07M
}
qpdf::impl::Writer& qpdf::impl::Writer::write_no_qdf<char const (&) [4]>(char const (&) [4])
Line
Count
Source
1310
160k
{
1311
160k
    if (!cfg.qdf()) {
1312
141k
        pipeline->write(std::forward<Args>(args)...);
1313
141k
    }
1314
160k
    return *this;
1315
160k
}
1316
1317
void
1318
impl::Writer::adjustAESStreamLength(size_t& length)
1319
364k
{
1320
364k
    if (encryption && !cur_data_key.empty() && cfg.encrypt_use_aes()) {
1321
        // Stream length will be padded with 1 to 16 bytes to end up as a multiple of 16.  It will
1322
        // also be prepended by 16 bits of random data.
1323
116k
        length += 32 - (length & 0xf);
1324
116k
    }
1325
364k
}
1326
1327
impl::Writer&
1328
impl::Writer::write_encrypted(std::string_view str)
1329
362k
{
1330
362k
    if (!(encryption && !cur_data_key.empty())) {
1331
178k
        write(str);
1332
184k
    } else if (cfg.encrypt_use_aes()) {
1333
115k
        write(pl::pipe<Pl_AES_PDF>(str, true, cur_data_key));
1334
115k
    } else {
1335
68.5k
        write(pl::pipe<Pl_RC4>(str, cur_data_key));
1336
68.5k
    }
1337
1338
362k
    return *this;
1339
362k
}
1340
1341
void
1342
impl::Writer::computeDeterministicIDData()
1343
34.0k
{
1344
34.0k
    if (!id2.empty()) {
1345
        // Can't happen in the code
1346
0
        throw std::logic_error(
1347
0
            "Deterministic ID computation enabled after ID generation has already occurred.");
1348
0
    }
1349
34.0k
    qpdf_assert_debug(deterministic_id_data.empty());
1350
34.0k
    deterministic_id_data = pipeline_stack.hex_digest();
1351
34.0k
}
1352
1353
int
1354
impl::Writer::openObject(int objid)
1355
1.28M
{
1356
1.28M
    if (objid == 0) {
1357
16.8k
        objid = next_objid++;
1358
16.8k
    }
1359
1.28M
    new_obj[objid].xref = QPDFXRefEntry(pipeline->getCount());
1360
1.28M
    write(objid).write(" 0 obj\n");
1361
1.28M
    return objid;
1362
1.28M
}
1363
1364
void
1365
impl::Writer::closeObject(int objid)
1366
1.28M
{
1367
    // Write a newline before endobj as it makes the file easier to repair.
1368
1.28M
    write("\nendobj\n").write_qdf("\n");
1369
1.28M
    auto& no = new_obj[objid];
1370
1.28M
    no.length = pipeline->getCount() - no.xref.getOffset();
1371
1.28M
}
1372
1373
void
1374
impl::Writer::assignCompressedObjectNumbers(QPDFObjGen og)
1375
430k
{
1376
430k
    int objid = og.getObj();
1377
430k
    if (og.getGen() != 0 || !object_stream_to_objects.contains(objid)) {
1378
        // This is not an object stream.
1379
400k
        return;
1380
400k
    }
1381
1382
    // Reserve numbers for the objects that belong to this object stream.
1383
295k
    for (auto const& iter: object_stream_to_objects[objid]) {
1384
295k
        obj[iter].renumber = next_objid++;
1385
295k
    }
1386
30.3k
}
1387
1388
void
1389
impl::Writer::enqueue(QPDFObjectHandle const& object)
1390
67.2M
{
1391
67.2M
    if (object.indirect()) {
1392
1.70M
        util::assertion(
1393
            // This owner check can only be done for indirect objects. It is possible for a direct
1394
            // object to have an owning QPDF that is from another file if a direct QPDFObjectHandle
1395
            // from one file was insert into another file without copying. Doing that is safe even
1396
            // if the original QPDF gets destroyed, which just disconnects the QPDFObjectHandle from
1397
            // its owner.
1398
1.70M
            object.qpdf() == &qpdf,
1399
1.70M
            "QPDFObjectHandle from different QPDF found while writing.  "
1400
1.70M
            "Use QPDF::copyForeignObject to add objects from another file." //
1401
1.70M
        );
1402
1403
1.70M
        if (cfg.qdf() && object.isStreamOfType("/XRef")) {
1404
            // As a special case, do not output any extraneous XRef streams in QDF mode. Doing so
1405
            // will confuse fix-qdf, which expects to see only one XRef stream at the end of the
1406
            // file. This case can occur when creating a QDF from a file with object streams when
1407
            // preserving unreferenced objects since the old cross reference streams are not
1408
            // actually referenced by object number.
1409
1.14k
            return;
1410
1.14k
        }
1411
1412
1.70M
        QPDFObjGen og = object.getObjGen();
1413
1.70M
        auto& o = obj[og];
1414
1415
1.70M
        if (o.renumber == 0) {
1416
772k
            if (o.object_stream > 0) {
1417
                // This is in an object stream.  Don't process it here.  Instead, enqueue the object
1418
                // stream.  Object streams always have generation 0.
1419
                // Detect loops by storing invalid object ID -1, which will get overwritten later.
1420
7.47k
                o.renumber = -1;
1421
7.47k
                enqueue(qpdf.getObject(o.object_stream, 0));
1422
765k
            } else {
1423
765k
                object_queue.emplace_back(object);
1424
765k
                o.renumber = next_objid++;
1425
1426
765k
                if (og.getGen() == 0 && object_stream_to_objects.contains(og.getObj())) {
1427
                    // For linearized files, uncompressed objects go at end, and we take care of
1428
                    // assigning numbers to them elsewhere.
1429
29.7k
                    if (!cfg.linearize()) {
1430
6.22k
                        assignCompressedObjectNumbers(og);
1431
6.22k
                    }
1432
735k
                } else if (!cfg.direct_stream_lengths() && object.isStream()) {
1433
                    // reserve next object ID for length
1434
49.1k
                    ++next_objid;
1435
49.1k
                }
1436
765k
            }
1437
772k
        }
1438
1.70M
        return;
1439
1.70M
    }
1440
1441
65.5M
    if (cfg.linearize()) {
1442
477
        return;
1443
477
    }
1444
1445
65.5M
    if (Array array = object) {
1446
48.1M
        for (auto& item: array) {
1447
48.1M
            enqueue(item);
1448
48.1M
        }
1449
2.39M
        return;
1450
2.39M
    }
1451
1452
63.2M
    for (auto const& item: Dictionary(object)) {
1453
8.55M
        if (!item.second.null()) {
1454
5.94M
            enqueue(item.second);
1455
5.94M
        }
1456
8.55M
    }
1457
63.2M
}
1458
1459
void
1460
impl::Writer::unparseChild(QPDFObjectHandle const& child, size_t level, int flags)
1461
21.3M
{
1462
21.3M
    if (!cfg.linearize()) {
1463
12.7M
        enqueue(child);
1464
12.7M
    }
1465
21.3M
    if (child.indirect()) {
1466
2.05M
        write(obj[child].renumber).write(" 0 R");
1467
19.2M
    } else {
1468
19.2M
        unparseObject(child, level, flags);
1469
19.2M
    }
1470
21.3M
}
1471
1472
void
1473
impl::Writer::writeTrailer(
1474
    trailer_e which, int size, bool xref_stream, qpdf_offset_t prev, int linearization_pass)
1475
160k
{
1476
160k
    auto trailer = trimmed_trailer();
1477
160k
    if (xref_stream) {
1478
60.4k
        cur_data_key.clear();
1479
99.5k
    } else {
1480
99.5k
        write("trailer <<");
1481
99.5k
    }
1482
160k
    write_qdf("\n");
1483
160k
    if (which == t_lin_second) {
1484
61.2k
        write(" /Size ").write(size);
1485
98.7k
    } else {
1486
206k
        for (auto const& [key, value]: trailer) {
1487
206k
            if (value.null()) {
1488
42.4k
                continue;
1489
42.4k
            }
1490
164k
            write_qdf("  ").write_no_qdf(" ").write_name(key).write(" ");
1491
164k
            if (key == "/Size") {
1492
17.0k
                write(size);
1493
17.0k
                if (which == t_lin_first) {
1494
11.1k
                    write(" /Prev ");
1495
11.1k
                    qpdf_offset_t pos = pipeline->getCount();
1496
11.1k
                    write(prev).write(QIntC::to_size(pos - pipeline->getCount() + 21), ' ');
1497
11.1k
                }
1498
147k
            } else {
1499
147k
                unparseChild(value, 1, 0);
1500
147k
            }
1501
164k
            write_qdf("\n");
1502
164k
        }
1503
98.7k
    }
1504
1505
    // Write ID
1506
160k
    write_qdf(" ").write(" /ID [");
1507
160k
    if (linearization_pass == 1) {
1508
63.0k
        std::string original_id1 = getOriginalID1();
1509
63.0k
        if (original_id1.empty()) {
1510
58.7k
            write("<00000000000000000000000000000000>");
1511
58.7k
        } else {
1512
            // Write a string of zeroes equal in length to the representation of the original ID.
1513
            // While writing the original ID would have the same number of bytes, it would cause a
1514
            // change to the deterministic ID generated by older versions of the software that
1515
            // hard-coded the length of the ID to 16 bytes.
1516
4.28k
            size_t len = QPDF_String(original_id1).unparse(true).length() - 2;
1517
4.28k
            write("<").write(len, '0').write(">");
1518
4.28k
        }
1519
63.0k
        write("<00000000000000000000000000000000>");
1520
96.9k
    } else {
1521
96.9k
        if (linearization_pass == 0 && cfg.deterministic_id()) {
1522
18.8k
            computeDeterministicIDData();
1523
18.8k
        }
1524
96.9k
        generateID(encryption.get());
1525
96.9k
        write_string(id1, true).write_string(id2, true);
1526
96.9k
    }
1527
160k
    write("]");
1528
1529
160k
    if (which != t_lin_second) {
1530
        // Write reference to encryption dictionary
1531
98.7k
        if (encryption) {
1532
48.7k
            write(" /Encrypt ").write(encryption_dict_objid).write(" 0 R");
1533
48.7k
        }
1534
98.7k
    }
1535
1536
160k
    write_qdf("\n>>").write_no_qdf(" >>");
1537
160k
}
1538
1539
bool
1540
impl::Writer::will_filter_stream(QPDFObjectHandle stream)
1541
92.9k
{
1542
92.9k
    std::string s;
1543
92.9k
    [[maybe_unused]] auto [filter, ignore1, ignore2] = will_filter_stream(stream, &s);
1544
92.9k
    return filter;
1545
92.9k
}
1546
1547
std::tuple<const bool, const bool, const bool>
1548
impl::Writer::will_filter_stream(QPDFObjectHandle stream, std::string* stream_data)
1549
380k
{
1550
380k
    const bool is_root_metadata = stream.isRootMetadata();
1551
380k
    bool filter = false;
1552
380k
    auto decode_level = cfg.decode_level();
1553
380k
    int encode_flags = 0;
1554
380k
    Dictionary stream_dict = stream.getDict();
1555
1556
380k
    if (stream.getFilterOnWrite()) {
1557
316k
        filter = stream.isDataModified() || cfg.compress_streams() || decode_level != qpdf_dl_none;
1558
316k
        if (cfg.compress_streams()) {
1559
            // Don't filter if the stream is already compressed with FlateDecode. This way we don't
1560
            // make it worse if the original file used a better Flate algorithm, and we don't spend
1561
            // time and CPU cycles uncompressing and recompressing stuff. This can be overridden
1562
            // with setRecompressFlate(true).
1563
267k
            Name Filter = stream_dict["/Filter"];
1564
267k
            if (Filter && !cfg.recompress_flate() && !stream.isDataModified() &&
1565
78.4k
                (Filter == "/FlateDecode" || Filter == "/Fl")) {
1566
34.8k
                filter = false;
1567
34.8k
            }
1568
267k
        }
1569
316k
        if (is_root_metadata && (!encryption || !encryption->getEncryptMetadata())) {
1570
559
            filter = true;
1571
559
            decode_level = qpdf_dl_all;
1572
315k
        } else if (cfg.normalize_content() && normalized_streams.contains(stream)) {
1573
6.40k
            encode_flags = qpdf_ef_normalize;
1574
6.40k
            filter = true;
1575
309k
        } else if (filter && cfg.compress_streams()) {
1576
231k
            encode_flags = qpdf_ef_compress;
1577
231k
        }
1578
316k
    }
1579
1580
    // Disable compression for empty streams to improve compatibility
1581
380k
    if (Integer(stream_dict["/Length"]) == 0) {
1582
17.9k
        filter = true;
1583
17.9k
        encode_flags = 0;
1584
17.9k
    }
1585
1586
482k
    for (bool first_attempt: {true, false}) {
1587
482k
        auto pp_stream_data =
1588
482k
            stream_data ? pipeline_stack.activate(*stream_data) : pipeline_stack.activate(true);
1589
1590
482k
        try {
1591
482k
            if (stream.pipeStreamData(
1592
482k
                    pipeline,
1593
482k
                    filter ? encode_flags : 0,
1594
482k
                    filter ? decode_level : qpdf_dl_none,
1595
482k
                    false,
1596
482k
                    first_attempt)) {
1597
191k
                return {true, encode_flags & qpdf_ef_compress, is_root_metadata};
1598
191k
            }
1599
290k
            if (!filter) {
1600
188k
                break;
1601
188k
            }
1602
290k
        } catch (std::runtime_error& e) {
1603
393
            if (!(filter && first_attempt)) {
1604
132
                throw std::runtime_error(
1605
132
                    "error while getting stream data for " + stream.unparse() + ": " + e.what());
1606
132
            }
1607
261
            stream.warn("error while getting stream data: "s + e.what());
1608
261
            stream.warn("qpdf will attempt to write the damaged stream unchanged");
1609
261
        }
1610
        // Try again
1611
101k
        filter = false;
1612
101k
        stream.setFilterOnWrite(false);
1613
101k
        if (stream_data) {
1614
101k
            stream_data->clear();
1615
101k
        }
1616
101k
    }
1617
188k
    return {false, false, is_root_metadata};
1618
380k
}
1619
1620
void
1621
impl::Writer::unparseObject(
1622
    QPDFObjectHandle object, size_t level, int flags, size_t stream_length, bool compress)
1623
21.0M
{
1624
21.0M
    QPDFObjGen old_og = object.getObjGen();
1625
21.0M
    int child_flags = flags & ~f_stream;
1626
    // For non-qdf, "indent" and "indent_large" are a single space between tokens. For qdf, they
1627
    // include the preceding newline.
1628
21.0M
    std::string indent_large = " ";
1629
21.0M
    if (cfg.qdf()) {
1630
11.3M
        indent_large.append(2 * (level + 1), ' ');
1631
11.3M
        indent_large[0] = '\n';
1632
11.3M
    }
1633
21.0M
    std::string_view indent{indent_large.data(), cfg.qdf() ? indent_large.size() - 2 : 1};
1634
1635
21.0M
    if (auto const tc = object.getTypeCode(); tc == ::ot_array) {
1636
        // Note: PDF spec 1.4 implementation note 121 states that Acrobat requires a space after the
1637
        // [ in the /H key of the linearization parameter dictionary.  We'll do this unconditionally
1638
        // for all arrays because it looks nicer and doesn't make the files that much bigger.
1639
671k
        write("[");
1640
17.2M
        for (auto const& item: object.as_array()) {
1641
17.2M
            write(indent_large);
1642
17.2M
            unparseChild(item, level + 1, child_flags);
1643
17.2M
        }
1644
671k
        write(indent).write("]");
1645
20.3M
    } else if (tc == ::ot_dictionary) {
1646
        // Handle special cases for specific dictionaries.
1647
1648
1.37M
        if (old_og == root_og) {
1649
            // Extensions dictionaries.
1650
1651
            // We have one of several cases:
1652
            //
1653
            // * We need ADBE
1654
            //    - We already have Extensions
1655
            //       - If it has the right ADBE, preserve it
1656
            //       - Otherwise, replace ADBE
1657
            //    - We don't have Extensions: create one from scratch
1658
            // * We don't want ADBE
1659
            //    - We already have Extensions
1660
            //       - If it only has ADBE, remove it
1661
            //       - If it has other things, keep those and remove ADBE
1662
            //    - We have no extensions: no action required
1663
            //
1664
            // Before writing, we guarantee that /Extensions, if present, is direct through the ADBE
1665
            // dictionary, so we can modify in place.
1666
1667
98.2k
            auto extensions = object.getKey("/Extensions");
1668
98.2k
            const bool has_extensions = extensions.isDictionary();
1669
98.2k
            const bool need_extensions_adbe = final_extension_level > 0;
1670
1671
98.2k
            if (has_extensions || need_extensions_adbe) {
1672
                // Make a shallow copy of this object so we can modify it safely without affecting
1673
                // the original. This code has logic to skip certain keys in agreement with
1674
                // prepareFileForWrite and with skip_stream_parameters so that replacing them
1675
                // doesn't leave unreferenced objects in the output. We can use unsafeShallowCopy
1676
                // here because all we are doing is removing or replacing top-level keys.
1677
34.4k
                object = object.unsafeShallowCopy();
1678
34.4k
                if (!has_extensions) {
1679
29.8k
                    extensions = QPDFObjectHandle();
1680
29.8k
                }
1681
1682
34.4k
                const bool have_extensions_adbe = extensions && extensions.hasKey("/ADBE");
1683
34.4k
                const bool have_extensions_other =
1684
34.4k
                    extensions && extensions.getKeys().size() > (have_extensions_adbe ? 1u : 0u);
1685
1686
34.4k
                if (need_extensions_adbe) {
1687
31.6k
                    if (!(have_extensions_other || have_extensions_adbe)) {
1688
                        // We need Extensions and don't have it.  Create it here.
1689
29.9k
                        QTC::TC("qpdf", "QPDFWriter create Extensions", cfg.qdf() ? 0 : 1);
1690
29.9k
                        extensions = object.replaceKeyAndGetNew(
1691
29.9k
                            "/Extensions", QPDFObjectHandle::newDictionary());
1692
29.9k
                    }
1693
31.6k
                } else if (!have_extensions_other) {
1694
                    // We have Extensions dictionary and don't want one.
1695
1.23k
                    if (have_extensions_adbe) {
1696
1.07k
                        QTC::TC("qpdf", "QPDFWriter remove existing Extensions");
1697
1.07k
                        object.removeKey("/Extensions");
1698
1.07k
                        extensions = QPDFObjectHandle(); // uninitialized
1699
1.07k
                    }
1700
1.23k
                }
1701
1702
34.4k
                if (extensions) {
1703
33.3k
                    QTC::TC("qpdf", "QPDFWriter preserve Extensions");
1704
33.3k
                    QPDFObjectHandle adbe = extensions.getKey("/ADBE");
1705
33.3k
                    if (adbe.isDictionary() &&
1706
1.66k
                        adbe.getKey("/BaseVersion").isNameAndEquals("/" + final_pdf_version) &&
1707
1.06k
                        adbe.getKey("/ExtensionLevel").isInteger() &&
1708
1.05k
                        (adbe.getKey("/ExtensionLevel").getIntValue() == final_extension_level)) {
1709
32.5k
                    } else {
1710
32.5k
                        if (need_extensions_adbe) {
1711
30.8k
                            extensions.replaceKey(
1712
30.8k
                                "/ADBE",
1713
30.8k
                                QPDFObjectHandle::parse(
1714
30.8k
                                    "<< /BaseVersion /" + final_pdf_version + " /ExtensionLevel " +
1715
30.8k
                                    std::to_string(final_extension_level) + " >>"));
1716
30.8k
                        } else {
1717
1.68k
                            extensions.removeKey("/ADBE");
1718
1.68k
                        }
1719
32.5k
                    }
1720
33.3k
                }
1721
34.4k
            }
1722
98.2k
        }
1723
1724
        // Stream dictionaries.
1725
1726
1.37M
        if (flags & f_stream) {
1727
            // Suppress /Length since we will write it manually
1728
1729
            // Make a shallow copy of this object so we can modify it safely without affecting the
1730
            // original. This code has logic to skip certain keys in agreement with
1731
            // prepareFileForWrite and with skip_stream_parameters so that replacing them doesn't
1732
            // leave unreferenced objects in the output. We can use unsafeShallowCopy here because
1733
            // all we are doing is removing or replacing top-level keys.
1734
287k
            object = object.unsafeShallowCopy();
1735
1736
287k
            object.removeKey("/Length");
1737
1738
            // If /DecodeParms is an empty list, remove it.
1739
287k
            if (object.getKey("/DecodeParms").empty()) {
1740
276k
                object.removeKey("/DecodeParms");
1741
276k
            }
1742
1743
287k
            if (flags & f_filtered) {
1744
                // We will supply our own filter and decode parameters.
1745
143k
                object.removeKey("/Filter");
1746
143k
                object.removeKey("/DecodeParms");
1747
143k
            } else {
1748
                // Make sure, no matter what else we have, that we don't have /Crypt in the output
1749
                // filters.
1750
143k
                QPDFObjectHandle filter = object.getKey("/Filter");
1751
143k
                QPDFObjectHandle decode_parms = object.getKey("/DecodeParms");
1752
143k
                if (filter.isOrHasName("/Crypt")) {
1753
3.39k
                    if (filter.isName()) {
1754
352
                        object.removeKey("/Filter");
1755
352
                        object.removeKey("/DecodeParms");
1756
3.04k
                    } else {
1757
3.04k
                        int idx = 0;
1758
121k
                        for (auto const& item: filter.as_array()) {
1759
121k
                            if (item.isNameAndEquals("/Crypt")) {
1760
                                // If filter is an array, then the code in QPDF_Stream has already
1761
                                // verified that DecodeParms and Filters are arrays of the same
1762
                                // length, but if they weren't for some reason, eraseItem does type
1763
                                // and bounds checking. Fuzzing tells us that this can actually
1764
                                // happen.
1765
3.04k
                                filter.eraseItem(idx);
1766
3.04k
                                decode_parms.eraseItem(idx);
1767
3.04k
                                break;
1768
3.04k
                            }
1769
118k
                            ++idx;
1770
118k
                        }
1771
3.04k
                    }
1772
3.39k
                }
1773
143k
            }
1774
287k
        }
1775
1776
1.37M
        write("<<");
1777
1778
4.89M
        for (auto const& [key, value]: object.as_dictionary()) {
1779
4.89M
            if (!value.null()) {
1780
3.97M
                write(indent_large).write_name(key).write(" ");
1781
3.97M
                if (key == "/Contents" && object.isDictionaryOfType("/Sig") &&
1782
674
                    object.hasKey("/ByteRange")) {
1783
658
                    QTC::TC("qpdf", "QPDFWriter no encryption sig contents");
1784
658
                    unparseChild(value, level + 1, child_flags | f_hex_string | f_no_encryption);
1785
3.97M
                } else {
1786
3.97M
                    unparseChild(value, level + 1, child_flags);
1787
3.97M
                }
1788
3.97M
            }
1789
4.89M
        }
1790
1791
1.37M
        if (flags & f_stream) {
1792
285k
            write(indent_large).write("/Length ");
1793
1794
285k
            if (cfg.direct_stream_lengths()) {
1795
236k
                write(stream_length);
1796
236k
            } else {
1797
48.9k
                write(cur_stream_length_id).write(" 0 R");
1798
48.9k
            }
1799
285k
            if (compress && (flags & f_filtered)) {
1800
119k
                write(indent_large).write("/Filter /FlateDecode");
1801
119k
            }
1802
285k
        }
1803
1804
1.37M
        write(indent).write(">>");
1805
19.0M
    } else if (tc == ::ot_stream) {
1806
        // Write stream data to a buffer.
1807
287k
        if (!cfg.direct_stream_lengths()) {
1808
49.1k
            cur_stream_length_id = obj[old_og].renumber + 1;
1809
49.1k
        }
1810
1811
287k
        flags |= f_stream;
1812
287k
        std::string stream_data;
1813
287k
        auto [filter, compress_stream, is_root_metadata] = will_filter_stream(object, &stream_data);
1814
287k
        if (filter) {
1815
143k
            flags |= f_filtered;
1816
143k
        }
1817
287k
        QPDFObjectHandle stream_dict = object.getDict();
1818
1819
287k
        cur_stream_length = stream_data.size();
1820
287k
        if (is_root_metadata && encryption && !encryption->getEncryptMetadata()) {
1821
            // Don't encrypt stream data for the metadata stream
1822
0
            cur_data_key.clear();
1823
0
        }
1824
287k
        adjustAESStreamLength(cur_stream_length);
1825
287k
        unparseObject(stream_dict, 0, flags, cur_stream_length, compress_stream);
1826
287k
        char last_char = stream_data.empty() ? '\0' : stream_data.back();
1827
287k
        write("\nstream\n").write_encrypted(stream_data);
1828
287k
        added_newline = cfg.newline_before_endstream() || (cfg.qdf() && last_char != '\n');
1829
287k
        write(added_newline ? "\nendstream" : "endstream");
1830
18.7M
    } else if (tc == ::ot_string) {
1831
604k
        std::string val;
1832
604k
        if (encryption && !(flags & f_in_ostream) && !(flags & f_no_encryption) &&
1833
136k
            !cur_data_key.empty()) {
1834
96.4k
            val = object.getStringValue();
1835
96.4k
            if (cfg.encrypt_use_aes()) {
1836
59.7k
                Pl_Buffer bufpl("encrypted string");
1837
59.7k
                Pl_AES_PDF pl("aes encrypt string", &bufpl, true, cur_data_key);
1838
59.7k
                pl.writeString(val);
1839
59.7k
                pl.finish();
1840
59.7k
                val = QPDF_String(bufpl.getString()).unparse(true);
1841
59.7k
            } else {
1842
36.6k
                auto tmp_ph = QUtil::make_unique_cstr(val);
1843
36.6k
                char* tmp = tmp_ph.get();
1844
36.6k
                size_t vlen = val.length();
1845
36.6k
                RC4 rc4(
1846
36.6k
                    QUtil::unsigned_char_pointer(cur_data_key),
1847
36.6k
                    QIntC::to_int(cur_data_key.length()));
1848
36.6k
                auto data = QUtil::unsigned_char_pointer(tmp);
1849
36.6k
                rc4.process(data, vlen, data);
1850
36.6k
                val = QPDF_String(std::string(tmp, vlen)).unparse();
1851
36.6k
            }
1852
507k
        } else if (flags & f_hex_string) {
1853
719
            val = QPDF_String(object.getStringValue()).unparse(true);
1854
506k
        } else {
1855
506k
            val = object.unparseResolved();
1856
506k
        }
1857
604k
        write(val);
1858
18.1M
    } else {
1859
18.1M
        write(object.unparseResolved());
1860
18.1M
    }
1861
21.0M
}
1862
1863
void
1864
impl::Writer::writeObjectStreamOffsets(std::vector<qpdf_offset_t>& offsets, int first_obj)
1865
93.1k
{
1866
93.1k
    qpdf_assert_debug(first_obj > 0);
1867
93.1k
    bool is_first = true;
1868
93.1k
    auto id = std::to_string(first_obj) + ' ';
1869
958k
    for (auto& offset: offsets) {
1870
958k
        if (is_first) {
1871
93.1k
            is_first = false;
1872
865k
        } else {
1873
865k
            write_qdf("\n").write_no_qdf(" ");
1874
865k
        }
1875
958k
        write(id);
1876
958k
        util::increment(id, 1);
1877
958k
        write(offset);
1878
958k
    }
1879
93.1k
    write("\n");
1880
93.1k
}
1881
1882
void
1883
impl::Writer::writeObjectStream(QPDFObjectHandle object)
1884
46.5k
{
1885
    // Note: object might be null if this is a place-holder for an object stream that we are
1886
    // generating from scratch.
1887
1888
46.5k
    QPDFObjGen old_og = object.getObjGen();
1889
46.5k
    qpdf_assert_debug(old_og.getGen() == 0);
1890
46.5k
    int old_id = old_og.getObj();
1891
46.5k
    int new_stream_id = obj[old_og].renumber;
1892
1893
46.5k
    std::vector<qpdf_offset_t> offsets;
1894
46.5k
    qpdf_offset_t first = 0;
1895
1896
    // Generate stream itself.  We have to do this in two passes so we can calculate offsets in the
1897
    // first pass.
1898
46.5k
    std::string stream_buffer_pass1;
1899
46.5k
    std::string stream_buffer_pass2;
1900
46.5k
    int first_obj = -1;
1901
46.5k
    const bool compressed = cfg.compress_streams() && !cfg.qdf();
1902
46.5k
    {
1903
        // Pass 1
1904
46.5k
        auto pp_ostream_pass1 = pipeline_stack.activate(stream_buffer_pass1);
1905
1906
46.5k
        int count = -1;
1907
479k
        for (auto const& og: object_stream_to_objects[old_id]) {
1908
479k
            ++count;
1909
479k
            int new_o = obj[og].renumber;
1910
479k
            if (first_obj == -1) {
1911
46.5k
                first_obj = new_o;
1912
46.5k
            }
1913
479k
            if (cfg.qdf()) {
1914
46.1k
                write("%% Object stream: object ").write(new_o).write(", index ").write(count);
1915
46.1k
                if (!cfg.no_original_object_ids()) {
1916
46.1k
                    write("; original object ID: ").write(og.getObj());
1917
                    // For compatibility, only write the generation if non-zero.  While object
1918
                    // streams only allow objects with generation 0, if we are generating object
1919
                    // streams, the old object could have a non-zero generation.
1920
46.1k
                    if (og.getGen() != 0) {
1921
0
                        write(" ").write(og.getGen());
1922
0
                    }
1923
46.1k
                }
1924
46.1k
                write("\n");
1925
46.1k
            }
1926
1927
479k
            offsets.push_back(pipeline->getCount());
1928
            // To avoid double-counting objects being written in object streams for progress
1929
            // reporting, decrement in pass 1.
1930
479k
            indicateProgress(true, false);
1931
1932
479k
            QPDFObjectHandle obj_to_write = qpdf.getObject(og);
1933
479k
            if (obj_to_write.isStream()) {
1934
                // This condition occurred in a fuzz input. Ideally we should block it at parse
1935
                // time, but it's not clear to me how to construct a case for this.
1936
4
                obj_to_write.warn("stream found inside object stream; treating as null");
1937
4
                obj_to_write = QPDFObjectHandle::newNull();
1938
4
            }
1939
479k
            writeObject(obj_to_write, count);
1940
1941
479k
            new_obj[new_o].xref = QPDFXRefEntry(new_stream_id, count);
1942
479k
        }
1943
46.5k
    }
1944
46.5k
    {
1945
        // Adjust offsets to skip over comment before first object
1946
46.5k
        first = offsets.at(0);
1947
479k
        for (auto& iter: offsets) {
1948
479k
            iter -= first;
1949
479k
        }
1950
1951
        // Take one pass at writing pairs of numbers so we can get their size information
1952
46.5k
        {
1953
46.5k
            auto pp_discard = pipeline_stack.activate(true);
1954
46.5k
            writeObjectStreamOffsets(offsets, first_obj);
1955
46.5k
            first += pipeline->getCount();
1956
46.5k
        }
1957
1958
        // Set up a stream to write the stream data into a buffer.
1959
46.5k
        auto pp_ostream = pipeline_stack.activate(stream_buffer_pass2);
1960
1961
46.5k
        writeObjectStreamOffsets(offsets, first_obj);
1962
46.5k
        write(stream_buffer_pass1);
1963
46.5k
        stream_buffer_pass1.clear();
1964
46.5k
        stream_buffer_pass1.shrink_to_fit();
1965
46.5k
        if (compressed) {
1966
40.6k
            stream_buffer_pass2 = pl::pipe<Pl_Flate>(stream_buffer_pass2, Pl_Flate::a_deflate);
1967
40.6k
        }
1968
46.5k
    }
1969
1970
    // Write the object
1971
46.5k
    openObject(new_stream_id);
1972
46.5k
    setDataKey(new_stream_id);
1973
46.5k
    write("<<").write_qdf("\n ").write(" /Type /ObjStm").write_qdf("\n ");
1974
46.5k
    size_t length = stream_buffer_pass2.size();
1975
46.5k
    adjustAESStreamLength(length);
1976
46.5k
    write(" /Length ").write(length).write_qdf("\n ");
1977
46.5k
    if (compressed) {
1978
40.6k
        write(" /Filter /FlateDecode");
1979
40.6k
    }
1980
46.5k
    write(" /N ").write(offsets.size()).write_qdf("\n ").write(" /First ").write(first);
1981
46.5k
    if (!object.null()) {
1982
        // If the original object has an /Extends key, preserve it.
1983
3.59k
        QPDFObjectHandle dict = object.getDict();
1984
3.59k
        QPDFObjectHandle extends = dict.getKey("/Extends");
1985
3.59k
        if (extends.isIndirect()) {
1986
1.01k
            write_qdf("\n ").write(" /Extends ");
1987
1.01k
            unparseChild(extends, 1, f_in_ostream);
1988
1.01k
        }
1989
3.59k
    }
1990
46.5k
    write_qdf("\n").write_no_qdf(" ").write(">>\nstream\n").write_encrypted(stream_buffer_pass2);
1991
46.5k
    write(cfg.newline_before_endstream() ? "\nendstream" : "endstream");
1992
46.5k
    if (encryption) {
1993
10.2k
        cur_data_key.clear();
1994
10.2k
    }
1995
46.5k
    closeObject(new_stream_id);
1996
46.5k
}
1997
1998
void
1999
impl::Writer::writeObject(QPDFObjectHandle object, int object_stream_index)
2000
1.51M
{
2001
1.51M
    QPDFObjGen old_og = object.getObjGen();
2002
2003
1.51M
    if (object_stream_index == -1 && old_og.getGen() == 0 &&
2004
1.02M
        object_stream_to_objects.contains(old_og.getObj())) {
2005
46.5k
        writeObjectStream(object);
2006
46.5k
        return;
2007
46.5k
    }
2008
2009
1.46M
    indicateProgress(false, false);
2010
1.46M
    auto new_id = obj[old_og].renumber;
2011
1.46M
    if (cfg.qdf()) {
2012
213k
        if (page_object_to_seq.contains(old_og)) {
2013
23.4k
            write("%% Page ").write(page_object_to_seq[old_og]).write("\n");
2014
23.4k
        }
2015
213k
        if (contents_to_page_seq.contains(old_og)) {
2016
17.4k
            write("%% Contents for page ").write(contents_to_page_seq[old_og]).write("\n");
2017
17.4k
        }
2018
213k
    }
2019
1.46M
    if (object_stream_index == -1) {
2020
988k
        if (cfg.qdf() && !cfg.no_original_object_ids()) {
2021
166k
            write("%% Original object ID: ").write(object.getObjGen().unparse(' ')).write("\n");
2022
166k
        }
2023
988k
        openObject(new_id);
2024
988k
        setDataKey(new_id);
2025
988k
        unparseObject(object, 0, 0);
2026
988k
        cur_data_key.clear();
2027
988k
        closeObject(new_id);
2028
988k
    } else {
2029
479k
        unparseObject(object, 0, f_in_ostream);
2030
479k
        write("\n");
2031
479k
    }
2032
2033
1.46M
    if (!cfg.direct_stream_lengths() && object.isStream()) {
2034
48.9k
        if (cfg.qdf()) {
2035
48.9k
            if (added_newline) {
2036
30.8k
                write("%QDF: ignore_newline\n");
2037
30.8k
            }
2038
48.9k
        }
2039
48.9k
        openObject(new_id + 1);
2040
48.9k
        write(cur_stream_length);
2041
48.9k
        closeObject(new_id + 1);
2042
48.9k
    }
2043
1.46M
}
2044
2045
std::string
2046
impl::Writer::getOriginalID1()
2047
133k
{
2048
133k
    if (String id0 = qpdf.getTrailer()["/ID"][0]) {
2049
9.41k
        return id0;
2050
9.41k
    }
2051
124k
    return "";
2052
133k
}
2053
2054
void
2055
impl::Writer::generateID(bool encrypted)
2056
133k
{
2057
    // Generate the ID lazily so that we can handle the user's preference to use static or
2058
    // deterministic ID generation.
2059
2060
133k
    if (!id2.empty()) {
2061
62.9k
        return;
2062
62.9k
    }
2063
2064
70.7k
    QPDFObjectHandle trailer = qpdf.getTrailer();
2065
2066
70.7k
    std::string result;
2067
2068
70.7k
    if (cfg.static_id()) {
2069
        // For test suite use only...
2070
36.5k
        static unsigned char tmp[] = {
2071
36.5k
            0x31,
2072
36.5k
            0x41,
2073
36.5k
            0x59,
2074
36.5k
            0x26,
2075
36.5k
            0x53,
2076
36.5k
            0x58,
2077
36.5k
            0x97,
2078
36.5k
            0x93,
2079
36.5k
            0x23,
2080
36.5k
            0x84,
2081
36.5k
            0x62,
2082
36.5k
            0x64,
2083
36.5k
            0x33,
2084
36.5k
            0x83,
2085
36.5k
            0x27,
2086
36.5k
            0x95,
2087
36.5k
            0x00};
2088
36.5k
        result = reinterpret_cast<char*>(tmp);
2089
36.5k
    } else {
2090
        // The PDF specification has guidelines for creating IDs, but it states clearly that the
2091
        // only thing that's really important is that it is very likely to be unique.  We can't
2092
        // really follow the guidelines in the spec exactly because we haven't written the file yet.
2093
        // This scheme should be fine though.  The deterministic ID case uses a digest of a
2094
        // sufficient portion of the file's contents such no two non-matching files would match in
2095
        // the subsets used for this computation.  Note that we explicitly omit the filename from
2096
        // the digest calculation for deterministic ID so that the same file converted with qpdf, in
2097
        // that case, would have the same ID regardless of the output file's name.
2098
2099
34.2k
        std::string seed;
2100
34.2k
        if (cfg.deterministic_id()) {
2101
34.2k
            if (encrypted) {
2102
172
                throw std::runtime_error(
2103
172
                    "QPDFWriter: unable to generated a deterministic ID because the file to be "
2104
172
                    "written is encrypted (even though the file may not require a password)");
2105
172
            }
2106
34.0k
            if (deterministic_id_data.empty()) {
2107
0
                throw std::logic_error(
2108
0
                    "INTERNAL ERROR: QPDFWriter::generateID has no data for deterministic ID");
2109
0
            }
2110
34.0k
            seed += deterministic_id_data;
2111
34.0k
        } else {
2112
0
            seed += std::to_string(QUtil::get_current_time());
2113
0
            seed += filename;
2114
0
            seed += " ";
2115
0
        }
2116
34.0k
        seed += " QPDF ";
2117
34.0k
        if (trailer.hasKey("/Info")) {
2118
20.2k
            for (auto const& item: trailer.getKey("/Info").as_dictionary()) {
2119
20.2k
                if (item.second.isString()) {
2120
5.02k
                    seed += " ";
2121
5.02k
                    seed += item.second.getStringValue();
2122
5.02k
                }
2123
20.2k
            }
2124
988
        }
2125
2126
34.0k
        MD5 md5;
2127
34.0k
        md5.encodeString(seed.c_str());
2128
34.0k
        MD5::Digest digest;
2129
34.0k
        md5.digest(digest);
2130
34.0k
        result = std::string(reinterpret_cast<char*>(digest), sizeof(MD5::Digest));
2131
34.0k
    }
2132
2133
    // If /ID already exists, follow the spec: use the original first word and generate a new second
2134
    // word.  Otherwise, we'll use the generated ID for both.
2135
2136
70.5k
    id2 = result;
2137
    // Note: keep /ID from old file even if --static-id was given.
2138
70.5k
    id1 = getOriginalID1();
2139
70.5k
    if (id1.empty()) {
2140
65.4k
        id1 = id2;
2141
65.4k
    }
2142
70.5k
}
2143
2144
void
2145
impl::Writer::initializeSpecialStreams()
2146
19.4k
{
2147
    // Mark all page content streams in case we are filtering or normalizing.
2148
19.4k
    int num = 0;
2149
24.0k
    for (auto& page: pages) {
2150
24.0k
        page_object_to_seq[page.getObjGen()] = ++num;
2151
24.0k
        QPDFObjectHandle contents = page.getKey("/Contents");
2152
24.0k
        std::vector<QPDFObjGen> contents_objects;
2153
24.0k
        if (contents.isArray()) {
2154
1.16k
            int n = static_cast<int>(contents.size());
2155
343k
            for (int i = 0; i < n; ++i) {
2156
342k
                contents_objects.push_back(contents.getArrayItem(i).getObjGen());
2157
342k
            }
2158
22.9k
        } else if (contents.isStream()) {
2159
3.54k
            contents_objects.push_back(contents.getObjGen());
2160
3.54k
        }
2161
2162
345k
        for (auto const& c: contents_objects) {
2163
345k
            contents_to_page_seq[c] = num;
2164
345k
            normalized_streams.insert(c);
2165
345k
        }
2166
24.0k
    }
2167
19.4k
}
2168
2169
void
2170
impl::Writer::preserveObjectStreams()
2171
38.6k
{
2172
38.6k
    auto const& xref = objects.xref_table();
2173
    // Our object_to_object_stream map has to map ObjGen -> ObjGen since we may be generating object
2174
    // streams out of old objects that have generation numbers greater than zero. However in an
2175
    // existing PDF, all object stream objects and all objects in them must have generation 0
2176
    // because the PDF spec does not provide any way to do otherwise. This code filters out objects
2177
    // that are not allowed to be in object streams. In addition to removing objects that were
2178
    // erroneously included in object streams in the source PDF, it also prevents unreferenced
2179
    // objects from being included.
2180
38.6k
    auto end = xref.cend();
2181
38.6k
    obj.streams_empty = true;
2182
38.6k
    if (cfg.preserve_unreferenced()) {
2183
0
        for (auto iter = xref.cbegin(); iter != end; ++iter) {
2184
0
            if (iter->second.getType() == 2) {
2185
                // Pdf contains object streams.
2186
0
                obj.streams_empty = false;
2187
0
                obj[iter->first].object_stream = iter->second.getObjStreamNumber();
2188
0
            }
2189
0
        }
2190
38.6k
    } else {
2191
        // Start by scanning for first compressed object in case we don't have any object streams to
2192
        // process.
2193
359k
        for (auto iter = xref.cbegin(); iter != end; ++iter) {
2194
325k
            if (iter->second.getType() == 2) {
2195
                // Pdf contains object streams.
2196
4.49k
                obj.streams_empty = false;
2197
4.49k
                auto eligible = objects.compressible_set();
2198
                // The object pointed to by iter may be a previous generation, in which case it is
2199
                // removed by compressible_set. We need to restart the loop (while the object
2200
                // table may contain multiple generations of an object).
2201
1.20M
                for (iter = xref.cbegin(); iter != end; ++iter) {
2202
1.20M
                    if (iter->second.getType() == 2) {
2203
1.11M
                        auto id = static_cast<size_t>(iter->first.getObj());
2204
1.11M
                        if (id < eligible.size() && eligible[id]) {
2205
139k
                            obj[iter->first].object_stream = iter->second.getObjStreamNumber();
2206
979k
                        } else {
2207
979k
                            QTC::TC("qpdf", "QPDFWriter exclude from object stream");
2208
979k
                        }
2209
1.11M
                    }
2210
1.20M
                }
2211
4.49k
                return;
2212
4.49k
            }
2213
325k
        }
2214
38.6k
    }
2215
38.6k
}
2216
2217
void
2218
impl::Writer::generateObjectStreams()
2219
20.2k
{
2220
    // Basic strategy: make a list of objects that can go into an object stream.  Then figure out
2221
    // how many object streams are needed so that we can distribute objects approximately evenly
2222
    // without having any object stream exceed 100 members.  We don't have to worry about linearized
2223
    // files here -- if the file is linearized, we take care of excluding things that aren't allowed
2224
    // here later.
2225
2226
    // This code doesn't do anything with /Extends.
2227
2228
20.2k
    auto eligible = objects.compressible_vector();
2229
20.2k
    size_t n_object_streams = (eligible.size() + 99U) / 100U;
2230
2231
20.2k
    initializeTables(2U * n_object_streams);
2232
20.2k
    if (n_object_streams == 0) {
2233
84
        obj.streams_empty = true;
2234
84
        return;
2235
84
    }
2236
20.1k
    size_t n_per = eligible.size() / n_object_streams;
2237
20.1k
    if (n_per * n_object_streams < eligible.size()) {
2238
245
        ++n_per;
2239
245
    }
2240
20.1k
    unsigned int n = 0;
2241
20.1k
    int cur_ostream = qpdf.newIndirectNull().getObjectID();
2242
220k
    for (auto const& item: eligible) {
2243
220k
        if (n == n_per) {
2244
1.18k
            n = 0;
2245
            // Construct a new null object as the "original" object stream.  The rest of the code
2246
            // knows that this means we're creating the object stream from scratch.
2247
1.18k
            cur_ostream = qpdf.newIndirectNull().getObjectID();
2248
1.18k
        }
2249
220k
        auto& o = obj[item];
2250
220k
        o.object_stream = cur_ostream;
2251
220k
        o.gen = item.getGen();
2252
220k
        ++n;
2253
220k
    }
2254
20.1k
}
2255
2256
Dictionary
2257
impl::Writer::trimmed_trailer()
2258
196k
{
2259
    // Remove keys from the trailer that necessarily have to be replaced when writing the file.
2260
2261
196k
    Dictionary trailer = qpdf.getTrailer().unsafeShallowCopy();
2262
2263
    // Remove encryption keys
2264
196k
    trailer.erase("/ID");
2265
196k
    trailer.erase("/Encrypt");
2266
2267
    // Remove modification information
2268
196k
    trailer.erase("/Prev");
2269
2270
    // Remove all trailer keys that potentially come from a cross-reference stream
2271
196k
    trailer.erase("/Index");
2272
196k
    trailer.erase("/W");
2273
196k
    trailer.erase("/Length");
2274
196k
    trailer.erase("/Filter");
2275
196k
    trailer.erase("/DecodeParms");
2276
196k
    trailer.erase("/Type");
2277
196k
    trailer.erase("/XRefStm");
2278
2279
196k
    return trailer;
2280
196k
}
2281
2282
// Make document extension level information direct as required by the spec.
2283
void
2284
impl::Writer::prepareFileForWrite()
2285
75.7k
{
2286
75.7k
    qpdf.fixDanglingReferences();
2287
75.7k
    auto root = qpdf.getRoot();
2288
75.7k
    auto oh = root.getKey("/Extensions");
2289
75.7k
    if (oh.isDictionary()) {
2290
3.79k
        const bool extensions_indirect = oh.isIndirect();
2291
3.79k
        if (extensions_indirect) {
2292
1.52k
            QTC::TC("qpdf", "QPDFWriter make Extensions direct");
2293
1.52k
            oh = root.replaceKeyAndGetNew("/Extensions", oh.shallowCopy());
2294
1.52k
        }
2295
3.79k
        if (oh.hasKey("/ADBE")) {
2296
2.40k
            auto adbe = oh.getKey("/ADBE");
2297
2.40k
            if (adbe.isIndirect()) {
2298
1.69k
                QTC::TC("qpdf", "QPDFWriter make ADBE direct", extensions_indirect ? 0 : 1);
2299
1.69k
                adbe.makeDirect();
2300
1.69k
                oh.replaceKey("/ADBE", adbe);
2301
1.69k
            }
2302
2.40k
        }
2303
3.79k
    }
2304
75.7k
}
2305
2306
void
2307
impl::Writer::initializeTables(size_t extra)
2308
76.1k
{
2309
76.1k
    auto size = objects.table_size() + 100u + extra;
2310
76.1k
    obj.resize(size);
2311
76.1k
    new_obj.resize(size);
2312
76.1k
}
2313
2314
void
2315
impl::Writer::doWriteSetup()
2316
76.3k
{
2317
76.3k
    if (did_write_setup) {
2318
0
        return;
2319
0
    }
2320
76.3k
    did_write_setup = true;
2321
2322
    // Do preliminary setup
2323
2324
76.3k
    if (cfg.linearize()) {
2325
39.6k
        cfg.qdf(false);
2326
39.6k
    }
2327
2328
76.3k
    if (cfg.pclm()) {
2329
0
        encryption = nullptr;
2330
0
    }
2331
2332
76.3k
    if (encryption) {
2333
        // Encryption has been explicitly set
2334
36.5k
        cfg.preserve_encryption(false);
2335
39.8k
    } else if (cfg.normalize_content() || cfg.pclm()) {
2336
        // Encryption makes looking at contents pretty useless.  If the user explicitly encrypted
2337
        // though, we still obey that.
2338
19.4k
        cfg.preserve_encryption(false);
2339
19.4k
    }
2340
2341
76.3k
    if (cfg.preserve_encryption()) {
2342
20.3k
        copyEncryptionParameters(qpdf);
2343
20.3k
    }
2344
2345
76.3k
    if (!cfg.forced_pdf_version().empty()) {
2346
0
        int major = 0;
2347
0
        int minor = 0;
2348
0
        parseVersion(cfg.forced_pdf_version(), major, minor);
2349
0
        disableIncompatibleEncryption(major, minor, cfg.forced_extension_level());
2350
0
        if (compareVersions(major, minor, 1, 5) < 0) {
2351
0
            cfg.object_streams(qpdf_o_disable);
2352
0
        }
2353
0
    }
2354
2355
76.3k
    if (cfg.qdf() || cfg.normalize_content()) {
2356
19.4k
        initializeSpecialStreams();
2357
19.4k
    }
2358
2359
76.3k
    switch (cfg.object_streams()) {
2360
17.2k
    case qpdf_o_disable:
2361
17.2k
        initializeTables();
2362
17.2k
        obj.streams_empty = true;
2363
17.2k
        break;
2364
2365
38.7k
    case qpdf_o_preserve:
2366
38.7k
        initializeTables();
2367
38.7k
        preserveObjectStreams();
2368
38.7k
        break;
2369
2370
20.2k
    case qpdf_o_generate:
2371
20.2k
        generateObjectStreams();
2372
20.2k
        break;
2373
76.3k
    }
2374
2375
76.1k
    if (!obj.streams_empty) {
2376
24.6k
        if (cfg.linearize()) {
2377
            // Page dictionaries are not allowed to be compressed objects.
2378
36.1k
            for (auto& page: pages) {
2379
36.1k
                if (obj[page].object_stream > 0) {
2380
30.2k
                    obj[page].object_stream = 0;
2381
30.2k
                }
2382
36.1k
            }
2383
22.6k
        }
2384
2385
24.6k
        if (cfg.linearize() || encryption) {
2386
            // The document catalog is not allowed to be compressed in linearized files either.
2387
            // It also appears that Adobe Reader 8.0.0 has a bug that prevents it from being able to
2388
            // handle encrypted files with compressed document catalogs, so we disable them in that
2389
            // case as well.
2390
22.6k
            if (obj[root_og].object_stream > 0) {
2391
17.4k
                obj[root_og].object_stream = 0;
2392
17.4k
            }
2393
22.6k
        }
2394
2395
        // Generate reverse mapping from object stream to objects
2396
14.8M
        obj.forEach([this](auto id, auto const& item) -> void {
2397
14.8M
            if (item.object_stream > 0) {
2398
310k
                auto& vec = object_stream_to_objects[item.object_stream];
2399
310k
                vec.emplace_back(id, item.gen);
2400
310k
                if (max_ostream_index < vec.size()) {
2401
118k
                    ++max_ostream_index;
2402
118k
                }
2403
310k
            }
2404
14.8M
        });
2405
24.6k
        --max_ostream_index;
2406
2407
24.6k
        if (object_stream_to_objects.empty()) {
2408
4.23k
            obj.streams_empty = true;
2409
20.3k
        } else {
2410
20.3k
            setMinimumPDFVersion("1.5");
2411
20.3k
        }
2412
24.6k
    }
2413
2414
76.1k
    setMinimumPDFVersion(qpdf.getPDFVersion(), qpdf.getExtensionLevel());
2415
76.1k
    final_pdf_version = min_pdf_version;
2416
76.1k
    final_extension_level = min_extension_level;
2417
76.1k
    if (!cfg.forced_pdf_version().empty()) {
2418
0
        final_pdf_version = cfg.forced_pdf_version();
2419
0
        final_extension_level = cfg.forced_extension_level();
2420
0
    }
2421
76.1k
}
2422
2423
void
2424
QPDFWriter::write()
2425
76.3k
{
2426
76.3k
    m->write();
2427
76.3k
}
2428
2429
void
2430
impl::Writer::write()
2431
76.3k
{
2432
76.3k
    doWriteSetup();
2433
2434
    // Set up progress reporting. For linearized files, we write two passes. events_expected is an
2435
    // approximation, but it's good enough for progress reporting, which is mostly a guess anyway.
2436
    // Count only objects that, when written, trigger an indicateProgress increment.
2437
76.3k
    size_t in_stream_objs = 0;
2438
76.3k
    for (auto const& kv: object_stream_to_objects) {
2439
35.5k
        in_stream_objs += kv.second.size();
2440
35.5k
    }
2441
76.3k
    const size_t expected =
2442
76.3k
        (qpdf.getObjectCount() - object_stream_to_objects.size() - in_stream_objs) *
2443
76.3k
        (cfg.linearize() ? 2 : 1);
2444
76.3k
    events_expected = std::max(
2445
76.3k
        1,
2446
76.3k
        util::fits<int>(expected) ? static_cast<int>(expected) : std::numeric_limits<int>::max());
2447
2448
76.3k
    prepareFileForWrite();
2449
2450
76.3k
    if (cfg.linearize()) {
2451
39.0k
        writeLinearized();
2452
39.0k
    } else {
2453
37.3k
        writeStandard();
2454
37.3k
    }
2455
2456
76.3k
    pipeline->finish();
2457
76.3k
    if (close_file) {
2458
0
        fclose(file);
2459
0
    }
2460
76.3k
    file = nullptr;
2461
76.3k
    if (buffer_pipeline) {
2462
0
        output_buffer = buffer_pipeline->getBuffer();
2463
0
        buffer_pipeline = nullptr;
2464
0
    }
2465
76.3k
    indicateProgress(false, true);
2466
76.3k
}
2467
2468
QPDFObjGen
2469
QPDFWriter::getRenumberedObjGen(QPDFObjGen og)
2470
0
{
2471
0
    return {m->obj[og].renumber, 0};
2472
0
}
2473
2474
std::map<QPDFObjGen, QPDFXRefEntry>
2475
QPDFWriter::getWrittenXRefTable()
2476
0
{
2477
0
    return m->getWrittenXRefTable();
2478
0
}
2479
2480
std::map<QPDFObjGen, QPDFXRefEntry>
2481
impl::Writer::getWrittenXRefTable()
2482
0
{
2483
0
    std::map<QPDFObjGen, QPDFXRefEntry> result;
2484
2485
0
    auto it = result.begin();
2486
0
    new_obj.forEach([&it, &result](auto id, auto const& item) -> void {
2487
0
        if (item.xref.getType() != 0) {
2488
0
            it = result.emplace_hint(it, QPDFObjGen(id, 0), item.xref);
2489
0
        }
2490
0
    });
2491
0
    return result;
2492
0
}
2493
2494
void
2495
impl::Writer::enqueuePart(std::vector<QPDFObjectHandle>& part)
2496
170k
{
2497
422k
    for (auto const& oh: part) {
2498
422k
        enqueue(oh);
2499
422k
    }
2500
170k
}
2501
2502
void
2503
impl::Writer::writeEncryptionDictionary()
2504
48.6k
{
2505
48.6k
    encryption_dict_objid = openObject(encryption_dict_objid);
2506
48.6k
    auto& enc = *encryption;
2507
48.6k
    auto const V = enc.getV();
2508
2509
48.6k
    write("<<");
2510
48.6k
    if (V >= 4) {
2511
31.8k
        write(" /CF << /StdCF << /AuthEvent /DocOpen /CFM ");
2512
31.8k
        write(cfg.encrypt_use_aes() ? (V < 5 ? "/AESV2" : "/AESV3") : "/V2");
2513
        // The PDF spec says the /Length key is optional, but the PDF previewer on some versions of
2514
        // MacOS won't open encrypted files without it.
2515
31.8k
        write(V < 5 ? " /Length 16 >> >>" : " /Length 32 >> >>");
2516
31.8k
        if (!encryption->getEncryptMetadata()) {
2517
0
            write(" /EncryptMetadata false");
2518
0
        }
2519
31.8k
    }
2520
48.6k
    write(" /Filter /Standard /Length ").write(enc.getLengthBytes() * 8);
2521
48.6k
    write(" /O ").write_string(enc.getO(), true);
2522
48.6k
    if (V >= 4) {
2523
31.8k
        write(" /OE ").write_string(enc.getOE(), true);
2524
31.8k
    }
2525
48.6k
    write(" /P ").write(enc.getP());
2526
48.6k
    if (V >= 5) {
2527
31.8k
        write(" /Perms ").write_string(enc.getPerms(), true);
2528
31.8k
    }
2529
48.6k
    write(" /R ").write(enc.getR());
2530
2531
48.6k
    if (V >= 4) {
2532
31.8k
        write(" /StmF /StdCF /StrF /StdCF");
2533
31.8k
    }
2534
48.6k
    write(" /U ").write_string(enc.getU(), true);
2535
48.6k
    if (V >= 4) {
2536
31.8k
        write(" /UE ").write_string(enc.getUE(), true);
2537
31.8k
    }
2538
48.6k
    write(" /V ").write(enc.getV()).write(" >>");
2539
48.6k
    closeObject(encryption_dict_objid);
2540
48.6k
}
2541
2542
std::string
2543
QPDFWriter::getFinalVersion()
2544
0
{
2545
0
    m->doWriteSetup();
2546
0
    return m->final_pdf_version;
2547
0
}
2548
2549
void
2550
impl::Writer::writeHeader()
2551
99.4k
{
2552
99.4k
    write("%PDF-").write(final_pdf_version);
2553
99.4k
    if (cfg.pclm()) {
2554
        // PCLm version
2555
0
        write("\n%PCLm 1.0\n");
2556
99.4k
    } else {
2557
        // This string of binary characters would not be valid UTF-8, so it really should be treated
2558
        // as binary.
2559
99.4k
        write("\n%\xbf\xf7\xa2\xfe\n");
2560
99.4k
    }
2561
99.4k
    write_qdf("%QDF-1.0\n\n");
2562
2563
    // Note: do not write extra header text here.  Linearized PDFs must include the entire
2564
    // linearization parameter dictionary within the first 1024 characters of the PDF file, so for
2565
    // linearized files, we have to write extra header text after the linearization parameter
2566
    // dictionary.
2567
99.4k
}
2568
2569
void
2570
impl::Writer::writeHintStream(int hint_id)
2571
30.6k
{
2572
30.6k
    std::string hint_buffer;
2573
30.6k
    int S = 0;
2574
30.6k
    int O = 0;
2575
30.6k
    bool compressed = cfg.compress_streams();
2576
30.6k
    lin.generateHintStream(new_obj, obj, hint_buffer, S, O, compressed);
2577
2578
30.6k
    openObject(hint_id);
2579
30.6k
    setDataKey(hint_id);
2580
2581
30.6k
    size_t hlen = hint_buffer.size();
2582
2583
30.6k
    write("<< ");
2584
30.6k
    if (compressed) {
2585
30.6k
        write("/Filter /FlateDecode ");
2586
30.6k
    }
2587
30.6k
    write("/S ").write(S);
2588
30.6k
    if (O) {
2589
673
        write(" /O ").write(O);
2590
673
    }
2591
30.6k
    adjustAESStreamLength(hlen);
2592
30.6k
    write(" /Length ").write(hlen);
2593
30.6k
    write(" >>\nstream\n").write_encrypted(hint_buffer);
2594
2595
30.6k
    if (encryption) {
2596
15.5k
        QTC::TC("qpdf", "QPDFWriter encrypted hint stream");
2597
15.5k
    }
2598
2599
30.6k
    write(hint_buffer.empty() || hint_buffer.back() != '\n' ? "\nendstream" : "endstream");
2600
30.6k
    closeObject(hint_id);
2601
30.6k
}
2602
2603
qpdf_offset_t
2604
impl::Writer::writeXRefTable(trailer_e which, int first, int last, int size)
2605
34.9k
{
2606
    // There are too many extra arguments to replace overloaded function with defaults in the header
2607
    // file...too much risk of leaving something off.
2608
34.9k
    return writeXRefTable(which, first, last, size, 0, false, 0, 0, 0, 0);
2609
34.9k
}
2610
2611
qpdf_offset_t
2612
impl::Writer::writeXRefTable(
2613
    trailer_e which,
2614
    int first,
2615
    int last,
2616
    int size,
2617
    qpdf_offset_t prev,
2618
    bool suppress_offsets,
2619
    int hint_id,
2620
    qpdf_offset_t hint_offset,
2621
    qpdf_offset_t hint_length,
2622
    int linearization_pass)
2623
99.5k
{
2624
99.5k
    write("xref\n").write(first).write(" ").write(last - first + 1);
2625
99.5k
    qpdf_offset_t space_before_zero = pipeline->getCount();
2626
99.5k
    write("\n");
2627
99.5k
    if (first == 0) {
2628
66.8k
        write("0000000000 65535 f \n");
2629
66.8k
        ++first;
2630
66.8k
    }
2631
944k
    for (int i = first; i <= last; ++i) {
2632
845k
        qpdf_offset_t offset = 0;
2633
845k
        if (!suppress_offsets) {
2634
674k
            offset = new_obj[i].xref.getOffset();
2635
674k
            if ((hint_id != 0) && (i != hint_id) && (offset >= hint_offset)) {
2636
91.2k
                offset += hint_length;
2637
91.2k
            }
2638
674k
        }
2639
845k
        write(QUtil::int_to_string(offset, 10)).write(" 00000 n \n");
2640
845k
    }
2641
99.5k
    writeTrailer(which, size, false, prev, linearization_pass);
2642
99.5k
    write("\n");
2643
99.5k
    return space_before_zero;
2644
99.5k
}
2645
2646
qpdf_offset_t
2647
impl::Writer::writeXRefStream(
2648
    int objid, int max_id, qpdf_offset_t max_offset, trailer_e which, int first, int last, int size)
2649
808
{
2650
    // There are too many extra arguments to replace overloaded function with defaults in the header
2651
    // file...too much risk of leaving something off.
2652
808
    return writeXRefStream(
2653
808
        objid, max_id, max_offset, which, first, last, size, 0, 0, 0, 0, false, 0);
2654
808
}
2655
2656
qpdf_offset_t
2657
impl::Writer::writeXRefStream(
2658
    int xref_id,
2659
    int max_id,
2660
    qpdf_offset_t max_offset,
2661
    trailer_e which,
2662
    int first,
2663
    int last,
2664
    int size,
2665
    qpdf_offset_t prev,
2666
    int hint_id,
2667
    qpdf_offset_t hint_offset,
2668
    qpdf_offset_t hint_length,
2669
    bool skip_compression,
2670
    int linearization_pass)
2671
60.4k
{
2672
60.4k
    qpdf_offset_t xref_offset = pipeline->getCount();
2673
60.4k
    qpdf_offset_t space_before_zero = xref_offset - 1;
2674
2675
    // field 1 contains offsets and object stream identifiers
2676
60.4k
    unsigned int f1_size = std::max(bytesNeeded(max_offset + hint_length), bytesNeeded(max_id));
2677
2678
    // field 2 contains object stream indices
2679
60.4k
    unsigned int f2_size = bytesNeeded(QIntC::to_longlong(max_ostream_index));
2680
2681
60.4k
    unsigned int esize = 1 + f1_size + f2_size;
2682
2683
    // Must store in xref table in advance of writing the actual data rather than waiting for
2684
    // openObject to do it.
2685
60.4k
    new_obj[xref_id].xref = QPDFXRefEntry(pipeline->getCount());
2686
2687
60.4k
    std::string xref_data;
2688
60.4k
    const bool compressed = cfg.compress_streams() && !cfg.qdf();
2689
60.4k
    {
2690
60.4k
        auto pp_xref = pipeline_stack.activate(xref_data);
2691
2692
1.04M
        for (int i = first; i <= last; ++i) {
2693
980k
            QPDFXRefEntry& e = new_obj[i].xref;
2694
980k
            switch (e.getType()) {
2695
234k
            case 0:
2696
234k
                writeBinary(0, 1);
2697
234k
                writeBinary(0, f1_size);
2698
234k
                writeBinary(0, f2_size);
2699
234k
                break;
2700
2701
351k
            case 1:
2702
351k
                {
2703
351k
                    qpdf_offset_t offset = e.getOffset();
2704
351k
                    if ((hint_id != 0) && (i != hint_id) && (offset >= hint_offset)) {
2705
95.6k
                        offset += hint_length;
2706
95.6k
                    }
2707
351k
                    writeBinary(1, 1);
2708
351k
                    writeBinary(QIntC::to_ulonglong(offset), f1_size);
2709
351k
                    writeBinary(0, f2_size);
2710
351k
                }
2711
351k
                break;
2712
2713
394k
            case 2:
2714
394k
                writeBinary(2, 1);
2715
394k
                writeBinary(QIntC::to_ulonglong(e.getObjStreamNumber()), f1_size);
2716
394k
                writeBinary(QIntC::to_ulonglong(e.getObjStreamIndex()), f2_size);
2717
394k
                break;
2718
2719
0
            default:
2720
0
                throw std::logic_error("invalid type writing xref stream");
2721
0
                break;
2722
980k
            }
2723
980k
        }
2724
60.4k
    }
2725
2726
60.4k
    if (compressed) {
2727
59.6k
        xref_data = pl::pipe<Pl_PNGFilter>(xref_data, Pl_PNGFilter::a_encode, esize);
2728
59.6k
        if (!skip_compression) {
2729
            // Write the stream dictionary for compression but don't actually compress.  This
2730
            // helps us with computation of padding for pass 1 of linearization.
2731
29.2k
            xref_data = pl::pipe<Pl_Flate>(xref_data, Pl_Flate::a_deflate);
2732
29.2k
        }
2733
59.6k
    }
2734
2735
60.4k
    openObject(xref_id);
2736
60.4k
    write("<<").write_qdf("\n ").write(" /Type /XRef").write_qdf("\n ");
2737
60.4k
    write(" /Length ").write(xref_data.size());
2738
60.4k
    if (compressed) {
2739
59.6k
        write_qdf("\n ").write(" /Filter /FlateDecode").write_qdf("\n ");
2740
59.6k
        write(" /DecodeParms << /Columns ").write(esize).write(" /Predictor 12 >>");
2741
59.6k
    }
2742
60.4k
    write_qdf("\n ").write(" /W [ 1 ").write(f1_size).write(" ").write(f2_size).write(" ]");
2743
60.4k
    if (!(first == 0 && last == (size - 1))) {
2744
30.4k
        write(" /Index [ ").write(first).write(" ").write(last - first + 1).write(" ]");
2745
30.4k
    }
2746
60.4k
    writeTrailer(which, size, true, prev, linearization_pass);
2747
60.4k
    write("\nstream\n").write(xref_data).write("\nendstream");
2748
60.4k
    closeObject(xref_id);
2749
60.4k
    return space_before_zero;
2750
60.4k
}
2751
2752
size_t
2753
impl::Writer::calculateXrefStreamPadding(qpdf_offset_t xref_bytes)
2754
30.4k
{
2755
    // This routine is called right after a linearization first pass xref stream has been written
2756
    // without compression.  Calculate the amount of padding that would be required in the worst
2757
    // case, assuming the number of uncompressed bytes remains the same. The worst case for zlib is
2758
    // that the output is larger than the input by 6 bytes plus 5 bytes per 16K, and then we'll add
2759
    // 10 extra bytes for number length increases.
2760
2761
30.4k
    return QIntC::to_size(16 + (5 * ((xref_bytes + 16383) / 16384)));
2762
30.4k
}
2763
2764
void
2765
impl::Writer::writeLinearized()
2766
39.0k
{
2767
    // Optimize file and enqueue objects in order
2768
2769
39.0k
    std::map<int, int> stream_cache;
2770
2771
158k
    auto skip_stream_parameters = [this, &stream_cache](QPDFObjectHandle& stream) {
2772
158k
        if (auto& result = stream_cache[stream.getObjectID()]) {
2773
65.9k
            return result;
2774
92.9k
        } else {
2775
92.9k
            return result = will_filter_stream(stream) ? 2 : 1;
2776
92.9k
        }
2777
158k
    };
2778
2779
39.0k
    lin.optimize(obj, skip_stream_parameters);
2780
2781
39.0k
    std::vector<QPDFObjectHandle> part4;
2782
39.0k
    std::vector<QPDFObjectHandle> part6;
2783
39.0k
    std::vector<QPDFObjectHandle> part7;
2784
39.0k
    std::vector<QPDFObjectHandle> part8;
2785
39.0k
    std::vector<QPDFObjectHandle> part9;
2786
39.0k
    lin.parts(obj, part4, part6, part7, part8, part9);
2787
2788
    // Object number sequence:
2789
    //
2790
    //  second half
2791
    //    second half uncompressed objects
2792
    //    second half xref stream, if any
2793
    //    second half compressed objects
2794
    //  first half
2795
    //    linearization dictionary
2796
    //    first half xref stream, if any
2797
    //    part 4 uncompresesd objects
2798
    //    encryption dictionary, if any
2799
    //    hint stream
2800
    //    part 6 uncompressed objects
2801
    //    first half compressed objects
2802
    //
2803
2804
    // Second half objects
2805
39.0k
    int second_half_uncompressed = QIntC::to_int(part7.size() + part8.size() + part9.size());
2806
39.0k
    int second_half_first_obj = 1;
2807
39.0k
    int after_second_half = 1 + second_half_uncompressed;
2808
39.0k
    next_objid = after_second_half;
2809
39.0k
    int second_half_xref = 0;
2810
39.0k
    bool need_xref_stream = !obj.streams_empty;
2811
39.0k
    if (need_xref_stream) {
2812
16.3k
        second_half_xref = next_objid++;
2813
16.3k
    }
2814
    // Assign numbers to all compressed objects in the second half.
2815
39.0k
    std::vector<QPDFObjectHandle>* vecs2[] = {&part7, &part8, &part9};
2816
142k
    for (int i = 0; i < 3; ++i) {
2817
166k
        for (auto const& oh: *vecs2[i]) {
2818
166k
            assignCompressedObjectNumbers(oh.getObjGen());
2819
166k
        }
2820
103k
    }
2821
39.0k
    int second_half_end = next_objid - 1;
2822
39.0k
    int second_trailer_size = next_objid;
2823
2824
    // First half objects
2825
39.0k
    int first_half_start = next_objid;
2826
39.0k
    int lindict_id = next_objid++;
2827
39.0k
    int first_half_xref = 0;
2828
39.0k
    if (need_xref_stream) {
2829
16.3k
        first_half_xref = next_objid++;
2830
16.3k
    }
2831
39.0k
    int part4_first_obj = next_objid;
2832
39.0k
    next_objid += QIntC::to_int(part4.size());
2833
39.0k
    int after_part4 = next_objid;
2834
39.0k
    if (encryption) {
2835
17.6k
        encryption_dict_objid = next_objid++;
2836
17.6k
    }
2837
39.0k
    int hint_id = next_objid++;
2838
39.0k
    int part6_first_obj = next_objid;
2839
39.0k
    next_objid += QIntC::to_int(part6.size());
2840
39.0k
    int after_part6 = next_objid;
2841
    // Assign numbers to all compressed objects in the first half
2842
39.0k
    std::vector<QPDFObjectHandle>* vecs1[] = {&part4, &part6};
2843
108k
    for (int i = 0; i < 2; ++i) {
2844
257k
        for (auto const& oh: *vecs1[i]) {
2845
257k
            assignCompressedObjectNumbers(oh.getObjGen());
2846
257k
        }
2847
69.1k
    }
2848
39.0k
    int first_half_end = next_objid - 1;
2849
39.0k
    int first_trailer_size = next_objid;
2850
2851
39.0k
    int part4_end_marker = part4.back().getObjectID();
2852
39.0k
    int part6_end_marker = part6.back().getObjectID();
2853
39.0k
    qpdf_offset_t space_before_zero = 0;
2854
39.0k
    qpdf_offset_t file_size = 0;
2855
39.0k
    qpdf_offset_t part6_end_offset = 0;
2856
39.0k
    qpdf_offset_t first_half_max_obj_offset = 0;
2857
39.0k
    qpdf_offset_t second_xref_offset = 0;
2858
39.0k
    qpdf_offset_t first_xref_end = 0;
2859
39.0k
    qpdf_offset_t second_xref_end = 0;
2860
2861
39.0k
    next_objid = part4_first_obj;
2862
39.0k
    enqueuePart(part4);
2863
39.0k
    if (next_objid != after_part4) {
2864
        // This can happen with very botched files as in the fuzzer test. There are likely some
2865
        // faulty assumptions in calculateLinearizationData
2866
41
        throw std::runtime_error("error encountered after writing part 4 of linearized data");
2867
41
    }
2868
38.9k
    next_objid = part6_first_obj;
2869
38.9k
    enqueuePart(part6);
2870
38.9k
    util::no_ci_rt_error_if(
2871
38.9k
        next_objid != after_part6, "error encountered after writing part 6 of linearized data" //
2872
38.9k
    );
2873
38.9k
    next_objid = second_half_first_obj;
2874
38.9k
    enqueuePart(part7);
2875
38.9k
    enqueuePart(part8);
2876
38.9k
    enqueuePart(part9);
2877
38.9k
    util::no_ci_rt_error_if(
2878
38.9k
        next_objid != after_second_half,
2879
38.9k
        "error encountered after writing part 9 of linearized data" //
2880
38.9k
    );
2881
2882
38.9k
    qpdf_offset_t hint_length = 0;
2883
38.9k
    std::string hint_buffer;
2884
2885
    // Write file in two passes.  Part numbers refer to PDF spec 1.4.
2886
2887
38.9k
    FILE* lin_pass1_file = nullptr;
2888
38.9k
    auto pp_pass1 = pipeline_stack.popper();
2889
38.9k
    auto pp_md5 = pipeline_stack.popper();
2890
63.0k
    for (int pass: {1, 2}) {
2891
63.0k
        if (pass == 1) {
2892
32.4k
            if (!cfg.linearize_pass1().empty()) {
2893
0
                lin_pass1_file = QUtil::safe_fopen(cfg.linearize_pass1().data(), "wb");
2894
0
                pipeline_stack.activate(
2895
0
                    pp_pass1,
2896
0
                    std::make_unique<Pl_StdioFile>("linearization pass1", lin_pass1_file));
2897
32.4k
            } else {
2898
32.4k
                pipeline_stack.activate(pp_pass1, true);
2899
32.4k
            }
2900
32.4k
            if (cfg.deterministic_id()) {
2901
15.9k
                pipeline_stack.activate_md5(pp_md5);
2902
15.9k
            }
2903
32.4k
        }
2904
2905
        // Part 1: header
2906
2907
63.0k
        writeHeader();
2908
2909
        // Part 2: linearization parameter dictionary.  Save enough space to write real dictionary.
2910
        // 200 characters is enough space if all numerical values in the parameter dictionary that
2911
        // contain offsets are 20 digits long plus a few extra characters for safety.  The entire
2912
        // linearization parameter dictionary must appear within the first 1024 characters of the
2913
        // file.
2914
2915
63.0k
        qpdf_offset_t pos = pipeline->getCount();
2916
63.0k
        openObject(lindict_id);
2917
63.0k
        write("<<");
2918
63.0k
        if (pass == 2) {
2919
30.6k
            write(" /Linearized 1 /L ").write(file_size + hint_length);
2920
            // Implementation note 121 states that a space is mandatory after this open bracket.
2921
30.6k
            write(" /H [ ").write(new_obj[hint_id].xref.getOffset()).write(" ");
2922
30.6k
            write(hint_length);
2923
30.6k
            write(" ] /O ").write(obj[pages.all().at(0)].renumber);
2924
30.6k
            write(" /E ").write(part6_end_offset + hint_length);
2925
30.6k
            write(" /N ").write(pages.size());
2926
30.6k
            write(" /T ").write(space_before_zero + hint_length);
2927
30.6k
        }
2928
63.0k
        write(" >>");
2929
63.0k
        closeObject(lindict_id);
2930
63.0k
        static int const pad = 200;
2931
63.0k
        write(QIntC::to_size(pos - pipeline->getCount() + pad), ' ').write("\n");
2932
2933
        // If the user supplied any additional header text, write it here after the linearization
2934
        // parameter dictionary.
2935
63.0k
        write(cfg.extra_header_text());
2936
2937
        // Part 3: first page cross reference table and trailer.
2938
2939
63.0k
        qpdf_offset_t first_xref_offset = pipeline->getCount();
2940
63.0k
        qpdf_offset_t hint_offset = 0;
2941
63.0k
        if (pass == 2) {
2942
30.6k
            hint_offset = new_obj[hint_id].xref.getOffset();
2943
30.6k
        }
2944
63.0k
        if (need_xref_stream) {
2945
            // Must pad here too.
2946
30.4k
            if (pass == 1) {
2947
                // Set first_half_max_obj_offset to a value large enough to force four bytes to be
2948
                // reserved for each file offset.  This would provide adequate space for the xref
2949
                // stream as long as the last object in page 1 starts with in the first 4 GB of the
2950
                // file, which is extremely likely.  In the second pass, we will know the actual
2951
                // value for this, but it's okay if it's smaller.
2952
15.7k
                first_half_max_obj_offset = 1 << 25;
2953
15.7k
            }
2954
30.4k
            pos = pipeline->getCount();
2955
30.4k
            writeXRefStream(
2956
30.4k
                first_half_xref,
2957
30.4k
                first_half_end,
2958
30.4k
                first_half_max_obj_offset,
2959
30.4k
                t_lin_first,
2960
30.4k
                first_half_start,
2961
30.4k
                first_half_end,
2962
30.4k
                first_trailer_size,
2963
30.4k
                hint_length + second_xref_offset,
2964
30.4k
                hint_id,
2965
30.4k
                hint_offset,
2966
30.4k
                hint_length,
2967
30.4k
                (pass == 1),
2968
30.4k
                pass);
2969
30.4k
            qpdf_offset_t endpos = pipeline->getCount();
2970
30.4k
            if (pass == 1) {
2971
                // Pad so we have enough room for the real xref stream.
2972
15.7k
                write(calculateXrefStreamPadding(endpos - pos), ' ');
2973
15.7k
                first_xref_end = pipeline->getCount();
2974
15.7k
            } else {
2975
                // Pad so that the next object starts at the same place as in pass 1.
2976
14.6k
                write(QIntC::to_size(first_xref_end - endpos), ' ');
2977
2978
14.6k
                if (pipeline->getCount() != first_xref_end) {
2979
0
                    throw std::logic_error(
2980
0
                        "insufficient padding for first pass xref stream; first_xref_end=" +
2981
0
                        std::to_string(first_xref_end) + "; endpos=" + std::to_string(endpos));
2982
0
                }
2983
14.6k
            }
2984
30.4k
            write("\n");
2985
32.6k
        } else {
2986
32.6k
            writeXRefTable(
2987
32.6k
                t_lin_first,
2988
32.6k
                first_half_start,
2989
32.6k
                first_half_end,
2990
32.6k
                first_trailer_size,
2991
32.6k
                hint_length + second_xref_offset,
2992
32.6k
                (pass == 1),
2993
32.6k
                hint_id,
2994
32.6k
                hint_offset,
2995
32.6k
                hint_length,
2996
32.6k
                pass);
2997
32.6k
            write("startxref\n0\n%%EOF\n");
2998
32.6k
        }
2999
3000
        // Parts 4 through 9
3001
3002
689k
        for (auto const& cur_object: object_queue) {
3003
689k
            if (cur_object.getObjectID() == part6_end_marker) {
3004
62.6k
                first_half_max_obj_offset = pipeline->getCount();
3005
62.6k
            }
3006
689k
            writeObject(cur_object);
3007
689k
            if (cur_object.getObjectID() == part4_end_marker) {
3008
62.9k
                if (encryption) {
3009
31.8k
                    writeEncryptionDictionary();
3010
31.8k
                }
3011
62.9k
                if (pass == 1) {
3012
32.2k
                    new_obj[hint_id].xref = QPDFXRefEntry(pipeline->getCount());
3013
32.2k
                } else {
3014
                    // Part 5: hint stream
3015
30.6k
                    write(hint_buffer);
3016
30.6k
                }
3017
62.9k
            }
3018
689k
            if (cur_object.getObjectID() == part6_end_marker) {
3019
62.1k
                part6_end_offset = pipeline->getCount();
3020
62.1k
            }
3021
689k
        }
3022
3023
        // Part 10: overflow hint stream -- not used
3024
3025
        // Part 11: main cross reference table and trailer
3026
3027
63.0k
        second_xref_offset = pipeline->getCount();
3028
63.0k
        if (need_xref_stream) {
3029
29.2k
            pos = pipeline->getCount();
3030
29.2k
            space_before_zero = writeXRefStream(
3031
29.2k
                second_half_xref,
3032
29.2k
                second_half_end,
3033
29.2k
                second_xref_offset,
3034
29.2k
                t_lin_second,
3035
29.2k
                0,
3036
29.2k
                second_half_end,
3037
29.2k
                second_trailer_size,
3038
29.2k
                0,
3039
29.2k
                0,
3040
29.2k
                0,
3041
29.2k
                0,
3042
29.2k
                (pass == 1),
3043
29.2k
                pass);
3044
29.2k
            qpdf_offset_t endpos = pipeline->getCount();
3045
3046
29.2k
            if (pass == 1) {
3047
                // Pad so we have enough room for the real xref stream.  See comments for previous
3048
                // xref stream on how we calculate the padding.
3049
14.6k
                write(calculateXrefStreamPadding(endpos - pos), ' ').write("\n");
3050
14.6k
                second_xref_end = pipeline->getCount();
3051
14.6k
            } else {
3052
                // Make the file size the same.
3053
14.5k
                auto padding =
3054
14.5k
                    QIntC::to_size(second_xref_end + hint_length - 1 - pipeline->getCount());
3055
14.5k
                write(padding, ' ').write("\n");
3056
3057
                // If this assertion fails, maybe we didn't have enough padding above.
3058
14.5k
                if (pipeline->getCount() != second_xref_end + hint_length) {
3059
0
                    throw std::logic_error(
3060
0
                        "count mismatch after xref stream; possible insufficient padding?");
3061
0
                }
3062
14.5k
            }
3063
33.8k
        } else {
3064
33.8k
            space_before_zero = writeXRefTable(
3065
33.8k
                t_lin_second, 0, second_half_end, second_trailer_size, 0, false, 0, 0, 0, pass);
3066
33.8k
        }
3067
63.0k
        write("startxref\n").write(first_xref_offset).write("\n%%EOF\n");
3068
3069
63.0k
        if (pass == 1) {
3070
30.6k
            if (cfg.deterministic_id()) {
3071
15.1k
                QTC::TC("qpdf", "QPDFWriter linearized deterministic ID", need_xref_stream ? 0 : 1);
3072
15.1k
                computeDeterministicIDData();
3073
15.1k
                pp_md5.pop();
3074
15.1k
            }
3075
3076
            // Close first pass pipeline
3077
30.6k
            file_size = pipeline->getCount();
3078
30.6k
            pp_pass1.pop();
3079
3080
            // Save hint offset since it will be set to zero by calling openObject.
3081
30.6k
            qpdf_offset_t hint_offset1 = new_obj[hint_id].xref.getOffset();
3082
3083
            // Write hint stream to a buffer
3084
30.6k
            {
3085
30.6k
                auto pp_hint = pipeline_stack.activate(hint_buffer);
3086
30.6k
                writeHintStream(hint_id);
3087
30.6k
            }
3088
30.6k
            hint_length = QIntC::to_offset(hint_buffer.size());
3089
3090
            // Restore hint offset
3091
30.6k
            new_obj[hint_id].xref = QPDFXRefEntry(hint_offset1);
3092
30.6k
            if (lin_pass1_file) {
3093
                // Write some debugging information
3094
0
                fprintf(
3095
0
                    lin_pass1_file, "%% hint_offset=%s\n", std::to_string(hint_offset1).c_str());
3096
0
                fprintf(lin_pass1_file, "%% hint_length=%s\n", std::to_string(hint_length).c_str());
3097
0
                fprintf(
3098
0
                    lin_pass1_file,
3099
0
                    "%% second_xref_offset=%s\n",
3100
0
                    std::to_string(second_xref_offset).c_str());
3101
0
                fprintf(
3102
0
                    lin_pass1_file,
3103
0
                    "%% second_xref_end=%s\n",
3104
0
                    std::to_string(second_xref_end).c_str());
3105
0
                fclose(lin_pass1_file);
3106
0
                lin_pass1_file = nullptr;
3107
0
            }
3108
30.6k
        }
3109
63.0k
    }
3110
38.9k
}
3111
3112
void
3113
impl::Writer::enqueueObjectsStandard()
3114
36.3k
{
3115
36.3k
    if (cfg.preserve_unreferenced()) {
3116
0
        for (auto const& oh: qpdf.getAllObjects()) {
3117
0
            enqueue(oh);
3118
0
        }
3119
0
    }
3120
3121
    // Put root first on queue.
3122
36.3k
    auto trailer = trimmed_trailer();
3123
36.3k
    enqueue(trailer["/Root"]);
3124
3125
    // Next place any other objects referenced from the trailer dictionary into the queue, handling
3126
    // direct objects recursively. Root is already there, so enqueuing it a second time is a no-op.
3127
69.7k
    for (auto& item: trailer) {
3128
69.7k
        if (!item.second.null()) {
3129
56.7k
            enqueue(item.second);
3130
56.7k
        }
3131
69.7k
    }
3132
36.3k
}
3133
3134
void
3135
impl::Writer::enqueueObjectsPCLm()
3136
0
{
3137
    // Image transform stream content for page strip images. Each of this new stream has to come
3138
    // after every page image strip written in the pclm file.
3139
0
    std::string image_transform_content = "q /image Do Q\n";
3140
3141
    // enqueue all pages first
3142
0
    for (auto& page: pages) {
3143
0
        enqueue(page);
3144
0
        enqueue(page["/Contents"]);
3145
3146
        // enqueue all the strips for each page
3147
0
        for (auto& image: Dictionary(page["/Resources"]["/XObject"])) {
3148
0
            if (!image.second.null()) {
3149
0
                enqueue(image.second);
3150
0
                enqueue(qpdf.newStream(image_transform_content));
3151
0
            }
3152
0
        }
3153
0
    }
3154
3155
0
    enqueue(trimmed_trailer()["/Root"]);
3156
0
}
3157
3158
void
3159
impl::Writer::indicateProgress(bool decrement, bool finished)
3160
2.01M
{
3161
2.01M
    if (decrement) {
3162
479k
        --events_seen;
3163
479k
        return;
3164
479k
    }
3165
3166
1.53M
    ++events_seen;
3167
3168
1.53M
    if (!progress_reporter.get()) {
3169
1.53M
        return;
3170
1.53M
    }
3171
3172
0
    if (finished || events_seen >= next_progress_report) {
3173
0
        int percentage =
3174
0
            (finished ? 100
3175
0
                 : next_progress_report == 0
3176
0
                 ? 0
3177
0
                 : std::min(99, 1 + ((100 * events_seen) / events_expected)));
3178
0
        if (cfg.linearize()) {
3179
            // When linearizing, we register a separate progress reporter with the linearization
3180
            // engine and scale its progress to the first half of overall progress. This logic moves
3181
            // the writer's internal progress to the second half. The division is not really 50%,
3182
            // but this still results in a meaningful and relatively steady advancement of progress.
3183
            // Otherwise, progress reporting doesn't start until after the linearization
3184
            // computations are done.
3185
0
            percentage = 50 + percentage / 2;
3186
0
        }
3187
0
        progress_reporter->reportProgress(percentage);
3188
0
    }
3189
0
    int increment = std::max(1, (events_expected / 100));
3190
0
    while (events_seen >= next_progress_report) {
3191
0
        next_progress_report += increment;
3192
0
    }
3193
0
}
3194
3195
void
3196
QPDFWriter::registerProgressReporter(std::shared_ptr<ProgressReporter> pr)
3197
0
{
3198
0
    m->progress_reporter = pr;
3199
    // When writing linearized files, linearization accounts for a significant fraction of the total
3200
    // time. Register a linearization progress callback, and then normalize its output to the first
3201
    // half of the overall progress. This is only used when we linearize. When linearizing,
3202
    // Writer::indicateProgress scales remaining progress to the second half.
3203
0
    m->lin.progress_callback([pr, last_reported = -1](int analysis_pct) mutable {
3204
0
        int p = analysis_pct / 2;
3205
0
        if (p != last_reported) {
3206
0
            pr->reportProgress(p);
3207
0
            last_reported = p;
3208
0
        }
3209
0
    });
3210
0
}
3211
3212
void
3213
impl::Writer::writeStandard()
3214
36.3k
{
3215
36.3k
    auto pp_md5 = pipeline_stack.popper();
3216
36.3k
    if (cfg.deterministic_id()) {
3217
19.2k
        pipeline_stack.activate_md5(pp_md5);
3218
19.2k
    }
3219
3220
    // Start writing
3221
3222
36.3k
    writeHeader();
3223
36.3k
    write(cfg.extra_header_text());
3224
3225
36.3k
    if (cfg.pclm()) {
3226
0
        enqueueObjectsPCLm();
3227
36.3k
    } else {
3228
36.3k
        enqueueObjectsStandard();
3229
36.3k
    }
3230
3231
    // Now start walking queue, outputting each object.
3232
381k
    while (object_queue_front < object_queue.size()) {
3233
345k
        QPDFObjectHandle cur_object = object_queue.at(object_queue_front);
3234
345k
        ++object_queue_front;
3235
345k
        writeObject(cur_object);
3236
345k
    }
3237
3238
    // Write out the encryption dictionary, if any
3239
36.3k
    if (encryption) {
3240
16.8k
        writeEncryptionDictionary();
3241
16.8k
    }
3242
3243
    // Now write out xref.  next_objid is now the number of objects.
3244
36.3k
    qpdf_offset_t xref_offset = pipeline->getCount();
3245
36.3k
    if (object_stream_to_objects.empty()) {
3246
        // Write regular cross-reference table
3247
34.9k
        writeXRefTable(t_normal, 0, next_objid - 1, next_objid);
3248
34.9k
    } else {
3249
        // Write cross-reference stream.
3250
1.45k
        int xref_id = next_objid++;
3251
1.45k
        writeXRefStream(xref_id, xref_id, xref_offset, t_normal, 0, next_objid - 1, next_objid);
3252
1.45k
    }
3253
36.3k
    write("startxref\n").write(xref_offset).write("\n%%EOF\n");
3254
3255
36.3k
    if (cfg.deterministic_id()) {
3256
18.8k
        QTC::TC(
3257
18.8k
            "qpdf",
3258
18.8k
            "QPDFWriter standard deterministic ID",
3259
18.8k
            object_stream_to_objects.empty() ? 0 : 1);
3260
18.8k
    }
3261
36.3k
}