Coverage Report

Created: 2026-09-28 10:59

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libreoffice/sax/source/expatwrap/xml2utf.cxx
Line
Count
Source
1
/* -*- Mode: C++; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */
2
/*
3
 * This file is part of the LibreOffice project.
4
 *
5
 * This Source Code Form is subject to the terms of the Mozilla Public
6
 * License, v. 2.0. If a copy of the MPL was not distributed with this
7
 * file, You can obtain one at http://mozilla.org/MPL/2.0/.
8
 *
9
 * This file incorporates work covered by the following license notice:
10
 *
11
 *   Licensed to the Apache Software Foundation (ASF) under one or more
12
 *   contributor license agreements. See the NOTICE file distributed
13
 *   with this work for additional information regarding copyright
14
 *   ownership. The ASF licenses this file to you under the Apache
15
 *   License, Version 2.0 (the "License"); you may not use this file
16
 *   except in compliance with the License. You may obtain a copy of
17
 *   the License at http://www.apache.org/licenses/LICENSE-2.0 .
18
 */
19
#include <string.h>
20
21
#include <algorithm>
22
23
#include <sal/types.h>
24
25
#include <comphelper/sequence.hxx>
26
#include <rtl/textenc.h>
27
#include <rtl/tencinfo.h>
28
#include <com/sun/star/io/NotConnectedException.hpp>
29
#include <com/sun/star/io/XInputStream.hpp>
30
#include <xml2utf.hxx>
31
#include <memory>
32
33
34
using namespace ::com::sun::star::uno;
35
using namespace ::com::sun::star::io;
36
37
38
namespace sax_expatwrap {
39
40
sal_Int32 XMLFile2UTFConverter::readAndConvert( Sequence<sal_Int8> &seq , sal_Int32 nMaxToRead )
41
366k
{
42
366k
    if( ! m_in.is() ) {
43
0
        throw NotConnectedException();
44
0
    }
45
366k
    if( ! m_bStarted ) {
46
        // it should be possible to find the encoding attribute
47
        // within the first 512 bytes == 128 chars in UCS-4
48
182k
        nMaxToRead = ::std::max( sal_Int32(512) , nMaxToRead );
49
182k
    }
50
51
366k
    sal_Int32 nRead;
52
366k
    Sequence< sal_Int8 > seqStart;
53
366k
    while( true )
54
366k
    {
55
366k
        nRead = m_in->readSomeBytes( seq , nMaxToRead );
56
57
366k
        if( nRead + seqStart.getLength())
58
191k
        {
59
            // if nRead is 0, the file is already eof.
60
191k
            if( ! m_bStarted && nRead )
61
176k
            {
62
                // ensure that enough data is available to parse encoding
63
176k
                if( seqStart.hasElements() )
64
36
                {
65
                  // prefix with what we had so far.
66
36
                    seq = comphelper::concatSequences(seqStart, seq);
67
36
                }
68
69
                // autodetection with the first bytes
70
176k
                if( ! isEncodingRecognizable( seq ) )
71
115
                {
72
                  // remember what we have so far.
73
115
                  seqStart = seq;
74
75
                  // read more !
76
115
                  continue;
77
115
                }
78
176k
                if( scanForEncoding( seq ) || !m_sEncoding.isEmpty() ) {
79
                    // initialize decoding
80
155k
                    initializeDecoding();
81
155k
                }
82
176k
                seqStart = Sequence < sal_Int8 > ();
83
176k
            }
84
85
            // do the encoding
86
191k
            if( m_pText2Unicode && m_pUnicode2Text &&
87
11.7k
                m_pText2Unicode->canContinue() ) {
88
89
11.0k
                Sequence<sal_Unicode> seqUnicode = m_pText2Unicode->convert( seq );
90
11.0k
                seq = m_pUnicode2Text->convert( seqUnicode.getConstArray(), seqUnicode.getLength() );
91
11.0k
            }
92
93
191k
            if( ! m_bStarted )
94
176k
            {
95
                // it must now be ensured, that no encoding attribute exist anymore
96
                // ( otherwise the expat-Parser will crash )
97
                // This must be done after decoding !
98
                // ( e.g. Files decoded in ucs-4 cannot be read properly )
99
176k
                m_bStarted = true;
100
176k
                removeEncoding( seq );
101
176k
            }
102
191k
            nRead = seq.getLength();
103
191k
        }
104
105
366k
        break;
106
366k
    }
107
366k
    return nRead;
108
366k
}
109
110
void XMLFile2UTFConverter::removeEncoding( Sequence<sal_Int8> &seq )
111
176k
{
112
176k
    const sal_Int8 *pSource = seq.getArray();
113
176k
    if (seq.getLength() < 5 || strncmp(reinterpret_cast<const char *>(pSource), "<?xml", 5))
114
22.4k
        return;
115
116
    // scan for encoding
117
153k
    OString str( reinterpret_cast<char const *>(pSource), seq.getLength() );
118
119
    // cut sequence to first line break
120
    // find first line break;
121
153k
    int nMax = str.indexOf( 10 );
122
153k
    if( nMax >= 0 )
123
147k
    {
124
147k
        str = str.copy( 0 , nMax );
125
147k
    }
126
127
153k
    int nFound = str.indexOf( " encoding" );
128
153k
    if( nFound < 0 )        return;
129
130
152k
    int nStop;
131
152k
    int nStart = str.indexOf( "\"" , nFound );
132
152k
    if( nStart < 0 || str.indexOf( "'" , nFound ) < nStart )
133
143k
    {
134
143k
        nStart = str.indexOf( "'" , nFound );
135
143k
        nStop  = str.indexOf( "'" , nStart +1 );
136
143k
    }
137
8.51k
    else
138
8.51k
    {
139
8.51k
        nStop  = str.indexOf( "\"" , nStart +1);
140
8.51k
    }
141
142
152k
    if( nStart >= 0 && nStop >= 0 && nStart+1 < nStop )
143
8.69k
    {
144
        // remove encoding tag from file
145
8.69k
        memmove(        &( seq.getArray()[nFound] ) ,
146
8.69k
                        &( seq.getArray()[nStop+1]) ,
147
8.69k
                        seq.getLength() - nStop -1);
148
8.69k
        seq.realloc( seq.getLength() - ( nStop+1 - nFound ) );
149
8.69k
    }
150
152k
}
151
152
// Checks, if enough data has been accumulated to recognize the encoding
153
bool XMLFile2UTFConverter::isEncodingRecognizable( const Sequence< sal_Int8 > &seq)
154
176k
{
155
176k
    const sal_Int8 *pSource = seq.getConstArray();
156
176k
    bool bCheckIfFirstClosingBracketExists = false;
157
158
176k
    if( seq.getLength() < 8 ) {
159
        // no recognition possible, when less than 8 bytes are available
160
35
        return false;
161
35
    }
162
163
176k
    if( ! strncmp( reinterpret_cast<const char *>(pSource), "<?xml", 5 ) ) {
164
        // scan if the <?xml tag finishes within this buffer
165
149k
        bCheckIfFirstClosingBracketExists = true;
166
149k
    }
167
26.6k
    else if( ('<' == pSource[0] || '<' == pSource[2] ) &&
168
20.5k
             ('?' == pSource[4] || '?' == pSource[6] ) )
169
393
    {
170
        // check for utf-16
171
393
        bCheckIfFirstClosingBracketExists = true;
172
393
    }
173
26.2k
    else if( ( '<' == pSource[1] || '<' == pSource[3] ) &&
174
6.65k
             ( '?' == pSource[5] || '?' == pSource[7] ) )
175
115
    {
176
        // check for
177
115
        bCheckIfFirstClosingBracketExists = true;
178
115
    }
179
180
176k
    if( bCheckIfFirstClosingBracketExists )
181
150k
    {
182
        // whole <?xml tag is valid
183
150k
        return std::find(seq.begin(), seq.end(), '>') != seq.end();
184
150k
    }
185
186
    // No <? tag in front, no need for a bigger buffer
187
26.1k
    return true;
188
176k
}
189
190
bool XMLFile2UTFConverter::scanForEncoding( Sequence< sal_Int8 > &seq )
191
176k
{
192
176k
    const sal_uInt8 *pSource = reinterpret_cast<const sal_uInt8*>( seq.getConstArray() );
193
176k
    bool bReturn = true;
194
195
176k
    if( seq.getLength() < 4 ) {
196
        // no recognition possible, when less than 4 bytes are available
197
0
        return false;
198
0
    }
199
200
    // first level : detect possible file formats
201
176k
    if (seq.getLength() >= 5 && !strncmp(reinterpret_cast<const char *>(pSource), "<?xml", 5)) {
202
        // scan for encoding
203
149k
        OString str( reinterpret_cast<const char *>(pSource), seq.getLength() );
204
205
        // cut sequence to first line break
206
        //find first line break;
207
149k
        int nMax = str.indexOf( 10 );
208
149k
        if( nMax >= 0 )
209
147k
        {
210
147k
            str = str.copy( 0 , nMax );
211
147k
        }
212
213
149k
        int nFound = str.indexOf( " encoding" );
214
149k
        if( nFound >= 0 ) {
215
148k
            int nStop;
216
148k
            int nStart = str.indexOf( "\"" , nFound );
217
148k
            if( nStart < 0 || str.indexOf( "'" , nFound ) < nStart )
218
139k
            {
219
139k
                nStart = str.indexOf( "'" , nFound );
220
139k
                nStop  = str.indexOf( "'" , nStart +1 );
221
139k
            }
222
8.45k
            else
223
8.45k
            {
224
8.45k
                nStop  = str.indexOf( "\"" , nStart +1);
225
8.45k
            }
226
148k
            if( nStart >= 0 && nStop >= 0 && nStart+1 < nStop )
227
8.64k
            {
228
                // encoding found finally
229
8.64k
                m_sEncoding = str.copy( nStart+1 , nStop - nStart - 1 );
230
8.64k
            }
231
148k
        }
232
149k
    }
233
26.5k
    else if( 0xFE == pSource[0] &&
234
211
             0xFF == pSource[1] ) {
235
        // UTF-16 big endian
236
        // conversion is done so that encoding information can be easily extracted
237
206
        m_sEncoding = "utf-16"_ostr;
238
206
    }
239
26.3k
    else if( 0xFF == pSource[0] &&
240
209
             0xFE == pSource[1] ) {
241
        // UTF-16 little endian
242
        // conversion is done so that encoding information can be easily extracted
243
205
        m_sEncoding = "utf-16"_ostr;
244
205
    }
245
26.1k
    else if( 0x00 == pSource[0] && 0x3c == pSource[1]  && 0x00 == pSource[2] && 0x3f == pSource[3] ) {
246
        // UTF-16 big endian without byte order mark (this is (strictly speaking) an error.)
247
        // The byte order mark is simply added
248
249
        // simply add the byte order mark !
250
27
        seq.realloc( seq.getLength() + 2 );
251
27
        memmove( &( seq.getArray()[2] ) , seq.getArray() , seq.getLength() - 2 );
252
27
        reinterpret_cast<sal_uInt8*>(seq.getArray())[0] = 0xFE;
253
27
        reinterpret_cast<sal_uInt8*>(seq.getArray())[1] = 0xFF;
254
255
27
        m_sEncoding = "utf-16"_ostr;
256
27
    }
257
26.1k
    else if( 0x3c == pSource[0] && 0x00 == pSource[1]  && 0x3f == pSource[2] && 0x00 == pSource[3] ) {
258
        // UTF-16 little endian without byte order mark (this is (strictly speaking) an error.)
259
        // The byte order mark is simply added
260
261
54
        seq.realloc( seq.getLength() + 2 );
262
54
        memmove( &( seq.getArray()[2] ) , seq.getArray() , seq.getLength() - 2 );
263
54
        reinterpret_cast<sal_uInt8*>(seq.getArray())[0] = 0xFF;
264
54
        reinterpret_cast<sal_uInt8*>(seq.getArray())[1] = 0xFE;
265
266
54
        m_sEncoding = "utf-16"_ostr;
267
54
    }
268
26.0k
    else if( 0xEF == pSource[0] &&
269
5.28k
             0xBB == pSource[1] &&
270
5.27k
             0xBF == pSource[2] )
271
5.27k
    {
272
        // UTF-8 BOM (byte order mark); signifies utf-8, and not byte order
273
        // The BOM is removed.
274
5.27k
        memmove( seq.getArray(), &( seq.getArray()[3] ), seq.getLength()-3 );
275
5.27k
        seq.realloc( seq.getLength() - 3 );
276
5.27k
        m_sEncoding = "utf-8"_ostr;
277
5.27k
    }
278
20.8k
    else if( 0x00 == pSource[0] && 0x00 == pSource[1]  && 0x00 == pSource[2] && 0x3c == pSource[3] ) {
279
        // UCS-4 big endian
280
3
        m_sEncoding = "ucs-4"_ostr;
281
3
    }
282
20.8k
    else if( 0x3c == pSource[0] && 0x00 == pSource[1]  && 0x00 == pSource[2] && 0x00 == pSource[3] ) {
283
        // UCS-4 little endian
284
5
        m_sEncoding = "ucs-4"_ostr;
285
5
    }
286
/* TODO: no need to test for the moment since we return sal_False like default case anyway
287
    else if( 0x4c == pSource[0] && 0x6f == pSource[1]  &&
288
             0xa7 == static_cast<unsigned char> (pSource[2]) &&
289
             0x94 == static_cast<unsigned char> (pSource[3]) ) {
290
        // EBCDIC
291
        bReturn = sal_False;   // must be extended
292
    }
293
*/
294
20.8k
    else {
295
        // other
296
        // UTF8 is directly recognized by the parser.
297
20.8k
        bReturn = false;
298
20.8k
    }
299
300
176k
    return bReturn;
301
176k
}
302
303
void XMLFile2UTFConverter::initializeDecoding()
304
155k
{
305
306
155k
    if( !m_sEncoding.isEmpty() )
307
14.4k
    {
308
14.4k
        rtl_TextEncoding encoding = rtl_getTextEncodingFromMimeCharset( m_sEncoding.getStr() );
309
14.4k
        if( encoding != RTL_TEXTENCODING_UTF8 )
310
9.07k
        {
311
9.07k
            m_pText2Unicode = std::make_unique<Text2UnicodeConverter>( m_sEncoding );
312
9.07k
            m_pUnicode2Text = std::make_unique<Unicode2TextConverter>( RTL_TEXTENCODING_UTF8 );
313
9.07k
        }
314
14.4k
    }
315
155k
}
316
317
318
// Text2UnicodeConverter
319
320
321
Text2UnicodeConverter::Text2UnicodeConverter( const OString &sEncoding )
322
9.07k
    : m_convText2Unicode(nullptr)
323
9.07k
    , m_contextText2Unicode(nullptr)
324
9.07k
{
325
9.07k
    rtl_TextEncoding encoding = rtl_getTextEncodingFromMimeCharset( sEncoding.getStr() );
326
9.07k
    if( RTL_TEXTENCODING_DONTKNOW == encoding )
327
657
    {
328
657
        m_bCanContinue = false;
329
657
        m_bInitialized = false;
330
657
    }
331
8.41k
    else
332
8.41k
    {
333
8.41k
        init( encoding );
334
8.41k
    }
335
9.07k
}
336
337
Text2UnicodeConverter::~Text2UnicodeConverter()
338
9.07k
{
339
9.07k
    if( m_bInitialized )
340
8.41k
    {
341
8.41k
        rtl_destroyTextToUnicodeContext( m_convText2Unicode , m_contextText2Unicode );
342
8.41k
        rtl_destroyUnicodeToTextConverter( m_convText2Unicode );
343
8.41k
    }
344
9.07k
}
345
346
void Text2UnicodeConverter::init( rtl_TextEncoding encoding )
347
8.41k
{
348
8.41k
    m_bCanContinue = true;
349
8.41k
    m_bInitialized = true;
350
351
8.41k
    m_convText2Unicode  = rtl_createTextToUnicodeConverter(encoding);
352
8.41k
    m_contextText2Unicode = rtl_createTextToUnicodeContext( m_convText2Unicode );
353
8.41k
}
354
355
356
Sequence<sal_Unicode> Text2UnicodeConverter::convert( const Sequence<sal_Int8> &seqText )
357
11.0k
{
358
11.0k
    sal_uInt32 uiInfo;
359
11.0k
    sal_Size nSrcCvtBytes   = 0;
360
11.0k
    sal_Size nTargetCount   = 0;
361
11.0k
    sal_Size nSourceCount   = 0;
362
363
    // the whole source size
364
11.0k
    sal_Int32   nSourceSize = seqText.getLength() + m_seqSource.getLength();
365
11.0k
    Sequence<sal_Unicode>   seqUnicode ( nSourceSize );
366
367
11.0k
    const sal_Int8 *pbSource = seqText.getConstArray();
368
11.0k
    std::unique_ptr<sal_Int8[]> pbTempMem;
369
370
11.0k
    if( m_seqSource.hasElements() ) {
371
        // put old rest and new byte sequence into one array
372
15
        pbTempMem.reset(new sal_Int8[ nSourceSize ]);
373
15
        memcpy( pbTempMem.get() , m_seqSource.getConstArray() , m_seqSource.getLength() );
374
15
        memcpy( &(pbTempMem[ m_seqSource.getLength() ]) , seqText.getConstArray() , seqText.getLength() );
375
15
        pbSource = pbTempMem.get();
376
377
        // set to zero again
378
15
        m_seqSource = Sequence< sal_Int8 >();
379
15
    }
380
381
11.0k
    while( true ) {
382
383
        /* All invalid characters are transformed to the unicode undefined char */
384
11.0k
        nTargetCount +=     rtl_convertTextToUnicode(
385
11.0k
                                    m_convText2Unicode,
386
11.0k
                                    m_contextText2Unicode,
387
11.0k
                                    reinterpret_cast<const char *>(&( pbSource[nSourceCount] )),
388
11.0k
                                    nSourceSize - nSourceCount ,
389
11.0k
                                    &( seqUnicode.getArray()[ nTargetCount ] ),
390
11.0k
                                    seqUnicode.getLength() - nTargetCount,
391
11.0k
                                    RTL_TEXTTOUNICODE_FLAGS_UNDEFINED_DEFAULT   |
392
11.0k
                                    RTL_TEXTTOUNICODE_FLAGS_MBUNDEFINED_DEFAULT |
393
11.0k
                                    RTL_TEXTTOUNICODE_FLAGS_INVALID_DEFAULT,
394
11.0k
                                    &uiInfo,
395
11.0k
                                    &nSrcCvtBytes );
396
11.0k
        nSourceCount += nSrcCvtBytes;
397
398
11.0k
        if( uiInfo & RTL_TEXTTOUNICODE_INFO_DESTBUFFERTOOSMALL ) {
399
            // save necessary bytes for next conversion
400
0
            seqUnicode.realloc( seqUnicode.getLength() * 2 );
401
0
            continue;
402
0
        }
403
11.0k
        break;
404
11.0k
    }
405
11.0k
    if( uiInfo & RTL_TEXTTOUNICODE_INFO_SRCBUFFERTOOSMALL ) {
406
406
        m_seqSource.realloc( nSourceSize - nSourceCount );
407
406
        memcpy( m_seqSource.getArray() , &(pbSource[nSourceCount]) , nSourceSize-nSourceCount );
408
406
    }
409
410
    // set to correct unicode size
411
11.0k
    seqUnicode.realloc( nTargetCount );
412
413
11.0k
    return seqUnicode;
414
11.0k
}
415
416
417
// Unicode2TextConverter
418
419
420
Unicode2TextConverter::Unicode2TextConverter( rtl_TextEncoding encoding )
421
9.07k
{
422
9.07k
    m_convUnicode2Text  = rtl_createUnicodeToTextConverter( encoding );
423
9.07k
    m_contextUnicode2Text = rtl_createUnicodeToTextContext( m_convUnicode2Text );
424
9.07k
}
425
426
427
Unicode2TextConverter::~Unicode2TextConverter()
428
9.07k
{
429
9.07k
    rtl_destroyUnicodeToTextContext( m_convUnicode2Text , m_contextUnicode2Text );
430
9.07k
    rtl_destroyUnicodeToTextConverter( m_convUnicode2Text );
431
9.07k
}
432
433
434
Sequence<sal_Int8> Unicode2TextConverter::convert(const sal_Unicode *puSource , sal_Int32 nSourceSize)
435
11.0k
{
436
11.0k
    std::unique_ptr<sal_Unicode[]> puTempMem;
437
438
11.0k
    if( m_seqSource.hasElements() ) {
439
        // For surrogates !
440
        // put old rest and new byte sequence into one array
441
        // In general when surrogates are used, they should be rarely
442
        // cut off between two convert()-calls. So this code is used
443
        // rarely and the extra copy is acceptable.
444
0
        puTempMem.reset(new sal_Unicode[ nSourceSize + m_seqSource.getLength()]);
445
0
        memcpy( puTempMem.get() ,
446
0
                m_seqSource.getConstArray() ,
447
0
                m_seqSource.getLength() * sizeof( sal_Unicode ) );
448
0
        memcpy(
449
0
            &(puTempMem[ m_seqSource.getLength() ]) ,
450
0
            puSource ,
451
0
            nSourceSize*sizeof( sal_Unicode ) );
452
0
        puSource = puTempMem.get();
453
0
        nSourceSize += m_seqSource.getLength();
454
455
0
        m_seqSource = Sequence< sal_Unicode > ();
456
0
    }
457
458
459
11.0k
    sal_Size nTargetCount = 0;
460
11.0k
    sal_Size nSourceCount = 0;
461
462
11.0k
    sal_uInt32 uiInfo;
463
11.0k
    sal_Size nSrcCvtChars;
464
465
    // take nSourceSize * 3 as preference
466
    // this is an upper boundary for converting to utf8,
467
    // which most often used as the target.
468
11.0k
    sal_Int32 nSeqSize =  nSourceSize * 3;
469
470
11.0k
    Sequence<sal_Int8>  seqText( nSeqSize );
471
11.0k
    char *pTarget = reinterpret_cast<char *>(seqText.getArray());
472
11.0k
    while( true ) {
473
474
11.0k
        nTargetCount += rtl_convertUnicodeToText(
475
11.0k
                                    m_convUnicode2Text,
476
11.0k
                                    m_contextUnicode2Text,
477
11.0k
                                    &( puSource[nSourceCount] ),
478
11.0k
                                    nSourceSize - nSourceCount ,
479
11.0k
                                    &( pTarget[nTargetCount] ),
480
11.0k
                                    nSeqSize - nTargetCount,
481
11.0k
                                    RTL_UNICODETOTEXT_FLAGS_UNDEFINED_DEFAULT |
482
11.0k
                                    RTL_UNICODETOTEXT_FLAGS_INVALID_DEFAULT ,
483
11.0k
                                    &uiInfo,
484
11.0k
                                    &nSrcCvtChars);
485
11.0k
        nSourceCount += nSrcCvtChars;
486
487
11.0k
        if( uiInfo & RTL_UNICODETOTEXT_INFO_DESTBUFFERTOSMALL ) {
488
0
            nSeqSize = nSeqSize *2;
489
0
            seqText.realloc( nSeqSize );  // double array size
490
0
            pTarget = reinterpret_cast<char *>(seqText.getArray());
491
0
            continue;
492
0
        }
493
11.0k
        break;
494
11.0k
    }
495
496
    // for surrogates
497
11.0k
    if( uiInfo & RTL_UNICODETOTEXT_INFO_SRCBUFFERTOSMALL ) {
498
0
        m_seqSource.realloc( nSourceSize - nSourceCount );
499
0
        memcpy( m_seqSource.getArray() ,
500
0
                &(puSource[nSourceCount]),
501
0
                (nSourceSize - nSourceCount) * sizeof( sal_Unicode ) );
502
0
    }
503
504
    // reduce the size of the buffer (fast, no copy necessary)
505
11.0k
    seqText.realloc( nTargetCount );
506
507
11.0k
    return seqText;
508
11.0k
}
509
510
}
511
512
/* vim:set shiftwidth=4 softtabstop=4 expandtab: */