Coverage Report

Created: 2026-09-28 10:59

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libreoffice/i18npool/source/search/textsearch.cxx
Line
Count
Source
1
/* -*- Mode: C++; tab-width: 4; indent-tabs-mode: nil; c-basic-offset: 4 -*- */
2
/*
3
 * This file is part of the LibreOffice project.
4
 *
5
 * This Source Code Form is subject to the terms of the Mozilla Public
6
 * License, v. 2.0. If a copy of the MPL was not distributed with this
7
 * file, You can obtain one at http://mozilla.org/MPL/2.0/.
8
 *
9
 * This file incorporates work covered by the following license notice:
10
 *
11
 *   Licensed to the Apache Software Foundation (ASF) under one or more
12
 *   contributor license agreements. See the NOTICE file distributed
13
 *   with this work for additional information regarding copyright
14
 *   ownership. The ASF licenses this file to you under the Apache
15
 *   License, Version 2.0 (the "License"); you may not use this file
16
 *   except in compliance with the License. You may obtain a copy of
17
 *   the License at http://www.apache.org/licenses/LICENSE-2.0 .
18
 */
19
20
#include "textsearch.hxx"
21
#include "levdis.hxx"
22
#include <com/sun/star/i18n/BreakIterator.hpp>
23
#include <com/sun/star/util/SearchAlgorithms2.hpp>
24
#include <com/sun/star/util/SearchFlags.hpp>
25
#include <com/sun/star/i18n/WordType.hpp>
26
#include <com/sun/star/i18n/ScriptType.hpp>
27
#include <com/sun/star/i18n/CharacterIteratorMode.hpp>
28
#include <com/sun/star/i18n/CharacterClassification.hpp>
29
#include <com/sun/star/i18n/KCharacterType.hpp>
30
#include <com/sun/star/i18n/Transliteration.hpp>
31
#include <cppuhelper/supportsservice.hxx>
32
#include <cppuhelper/weak.hxx>
33
#include <i18nutil/transliteration.hxx>
34
#include <o3tl/string_view.hxx>
35
#include <rtl/ustrbuf.hxx>
36
#include <sal/log.hxx>
37
38
#include <unicode/regex.h>
39
40
using namespace ::com::sun::star::util;
41
using namespace ::com::sun::star::uno;
42
using namespace ::com::sun::star::lang;
43
using namespace ::com::sun::star::i18n;
44
using namespace ::com::sun::star;
45
46
const TransliterationFlags COMPLEX_TRANS_MASK =
47
    TransliterationFlags::ignoreBaFa_ja_JP |
48
    TransliterationFlags::ignoreIterationMark_ja_JP |
49
    TransliterationFlags::ignoreTiJi_ja_JP |
50
    TransliterationFlags::ignoreHyuByu_ja_JP |
51
    TransliterationFlags::ignoreSeZe_ja_JP |
52
    TransliterationFlags::ignoreIandEfollowedByYa_ja_JP |
53
    TransliterationFlags::ignoreKiKuFollowedBySa_ja_JP |
54
    TransliterationFlags::ignoreProlongedSoundMark_ja_JP;
55
56
namespace
57
{
58
TransliterationFlags maskComplexTrans( TransliterationFlags n )
59
0
{
60
    // IGNORE_KANA and FULLWIDTH_HALFWIDTH are simple but need to take effect
61
    // in complex transliteration.
62
0
    return
63
0
        n & (COMPLEX_TRANS_MASK |                       // all set ignore bits
64
0
        TransliterationFlags::IGNORE_KANA |            // plus IGNORE_KANA bit
65
0
        TransliterationFlags::FULLWIDTH_HALFWIDTH);    // and the FULLWIDTH_HALFWIDTH value
66
0
}
67
68
bool isComplexTrans( TransliterationFlags n )
69
6
{
70
6
    return bool(n & COMPLEX_TRANS_MASK);
71
6
}
72
73
TransliterationFlags maskSimpleTrans( TransliterationFlags n )
74
12
{
75
12
    return n & ~COMPLEX_TRANS_MASK;
76
12
}
77
78
bool isSimpleTrans( TransliterationFlags n )
79
9
{
80
9
    return bool(maskSimpleTrans(n));
81
9
}
82
83
// Regex patterns are case sensitive.
84
TransliterationFlags maskSimpleRegexTrans( TransliterationFlags n )
85
0
{
86
0
    TransliterationFlags m = (n & TransliterationFlags::IGNORE_MASK) & ~TransliterationFlags::IGNORE_CASE;
87
0
    TransliterationFlags v = n & TransliterationFlags::NON_IGNORE_MASK;
88
0
    if (v == TransliterationFlags::UPPERCASE_LOWERCASE || v == TransliterationFlags::LOWERCASE_UPPERCASE)
89
0
        v = TransliterationFlags::NONE;
90
0
    return (m | v) & ~COMPLEX_TRANS_MASK;
91
0
}
92
93
bool isSimpleRegexTrans( TransliterationFlags n )
94
0
{
95
0
    return bool(maskSimpleRegexTrans(n));
96
0
}
97
98
bool isReplacePunctuation( std::u16string_view rStr )
99
0
{
100
0
    return rStr.find(u'\u2018') != std::u16string_view::npos ||
101
0
           rStr.find(u'\u2019') != std::u16string_view::npos ||
102
0
           rStr.find(u'\u201A') != std::u16string_view::npos ||
103
0
           rStr.find(u'\u201B') != std::u16string_view::npos ||
104
0
           rStr.find(u'\u201C') != std::u16string_view::npos ||
105
0
           rStr.find(u'\u201D') != std::u16string_view::npos ||
106
0
           rStr.find(u'\u201E') != std::u16string_view::npos ||
107
0
           rStr.find(u'\u201F') != std::u16string_view::npos;
108
0
}
109
110
OUString replacePunctuation( const OUString &rStr )
111
0
{
112
0
    return rStr.replace(u'\u2018', '\'')
113
0
               .replace(u'\u2019', '\'')
114
0
               .replace(u'\u201A', '\'')
115
0
               .replace(u'\u201B', '\'')
116
0
               .replace(u'\u201C', '"')
117
0
               .replace(u'\u201D', '"')
118
0
               .replace(u'\u201E', '"')
119
0
               .replace(u'\u201F', '"');
120
0
}
121
}
122
123
TextSearch::TextSearch(const Reference < XComponentContext > & rxContext)
124
3
        : m_xContext( rxContext )
125
3
{
126
3
    SearchOptions2 aOpt;
127
3
    aOpt.AlgorithmType2 = SearchAlgorithms2::ABSOLUTE;
128
3
    aOpt.algorithmType = SearchAlgorithms_ABSOLUTE;
129
3
    aOpt.searchFlag = SearchFlags::ALL_IGNORE_CASE;
130
    //aOpt.Locale = ???;
131
3
    setOptions( aOpt );
132
3
}
133
134
TextSearch::~TextSearch()
135
3
{
136
3
    pRegexMatcher.reset();
137
3
    pWLD.reset();
138
3
    pJumpTable.reset();
139
3
    pJumpTable2.reset();
140
3
}
141
142
void TextSearch::setOptions2( const SearchOptions2& rOptions )
143
6
{
144
6
    std::unique_lock g(m_aMutex);
145
146
6
    aSrchPara = rOptions;
147
148
6
    pRegexMatcher.reset();
149
6
    pWLD.reset();
150
6
    pJumpTable.reset();
151
6
    pJumpTable2.reset();
152
6
    maWildcardReversePattern.clear();
153
6
    maWildcardReversePattern2.clear();
154
6
    TransliterationFlags transliterateFlags = static_cast<TransliterationFlags>(aSrchPara.transliterateFlags);
155
6
    bSearchApostrophe = false;
156
6
    bool bReplaceApostrophe = false;
157
6
    if (aSrchPara.AlgorithmType2 == SearchAlgorithms2::REGEXP)
158
0
    {
159
        // RESrchPrepare will consider aSrchPara.transliterateFlags when
160
        // picking the actual regex pattern
161
        // (sSrchStr|sSrchStr2|SearchOptions2::searchString) and setting
162
        // case-insensitivity. Create transliteration instance, if any, without
163
        // ignore-case so later in TextSearch::searchForward() the string to
164
        // match is not case-altered, leave case-(in)sensitive to regex engine.
165
0
        transliterateFlags &= ~TransliterationFlags::IGNORE_CASE;
166
0
    }
167
6
    else if ( aSrchPara.searchString.indexOf('\'') > -1 || aSrchPara.searchString.indexOf('"') > -1 )
168
0
    {
169
0
        bSearchApostrophe = true;
170
0
        bReplaceApostrophe = isReplacePunctuation(aSrchPara.searchString);
171
0
    }
172
173
    // Create Transliteration class
174
6
    if( isSimpleTrans( transliterateFlags) )
175
3
    {
176
3
        if( !xTranslit.is() )
177
3
            xTranslit.set( Transliteration::create( m_xContext ) );
178
3
        xTranslit->loadModule(
179
3
             static_cast<TransliterationModules>(maskSimpleTrans(transliterateFlags)),
180
3
             aSrchPara.Locale);
181
3
    }
182
3
    else if( xTranslit.is() )
183
0
        xTranslit = nullptr;
184
185
    // Create Transliteration for 2<->1, 2<->2 transliteration
186
6
    if ( isComplexTrans( transliterateFlags) )
187
0
    {
188
0
        if( !xTranslit2.is() )
189
0
            xTranslit2.set( Transliteration::create( m_xContext ) );
190
        // Load transliteration module
191
0
        xTranslit2->loadModule(
192
0
             static_cast<TransliterationModules>(maskComplexTrans(transliterateFlags)),
193
0
             aSrchPara.Locale);
194
0
    }
195
196
6
    if ( !xBreak.is() )
197
3
        xBreak = css::i18n::BreakIterator::create( m_xContext );
198
199
6
    sSrchStr = aSrchPara.searchString;
200
201
    // Transliterate search string.
202
6
    if (aSrchPara.AlgorithmType2 == SearchAlgorithms2::REGEXP)
203
0
    {
204
0
        if (isSimpleRegexTrans(transliterateFlags))
205
0
        {
206
0
            if (maskSimpleRegexTrans(transliterateFlags) !=
207
0
                    maskSimpleTrans(transliterateFlags))
208
0
            {
209
0
                css::uno::Reference< XExtendedTransliteration > xTranslitPattern(
210
0
                         Transliteration::create( m_xContext ));
211
0
                if (xTranslitPattern.is())
212
0
                {
213
0
                    xTranslitPattern->loadModule(
214
0
                            static_cast<TransliterationModules>(maskSimpleRegexTrans(transliterateFlags)),
215
0
                            aSrchPara.Locale);
216
0
                    sSrchStr = xTranslitPattern->transliterateString2String(
217
0
                            aSrchPara.searchString, 0, aSrchPara.searchString.getLength());
218
0
                }
219
0
            }
220
0
            else
221
0
            {
222
0
                if (xTranslit.is())
223
0
                    sSrchStr = xTranslit->transliterateString2String(
224
0
                            aSrchPara.searchString, 0, aSrchPara.searchString.getLength());
225
0
            }
226
            // xTranslit2 complex transliterated sSrchStr2 is not used in
227
            // regex, see TextSearch::searchForward() and
228
            // TextSearch::searchBackward()
229
0
        }
230
0
    }
231
6
    else
232
6
    {
233
6
        if ( xTranslit.is() && isSimpleTrans(transliterateFlags) )
234
3
            sSrchStr = xTranslit->transliterateString2String(
235
3
                    aSrchPara.searchString, 0, aSrchPara.searchString.getLength());
236
237
6
        if ( xTranslit2.is() && isComplexTrans(transliterateFlags) )
238
0
            sSrchStr2 = xTranslit2->transliterateString2String(
239
0
                    aSrchPara.searchString, 0, aSrchPara.searchString.getLength());
240
6
    }
241
242
6
    if ( bReplaceApostrophe )
243
0
        sSrchStr = replacePunctuation(sSrchStr);
244
245
    // Take the new SearchOptions2::AlgorithmType2 field and ignore
246
    // SearchOptions::algorithmType
247
6
    switch( aSrchPara.AlgorithmType2)
248
6
    {
249
0
        case SearchAlgorithms2::REGEXP:
250
0
            fnForward = &TextSearch::RESrchFrwrd;
251
0
            fnBackward = &TextSearch::RESrchBkwrd;
252
0
            RESrchPrepare( aSrchPara);
253
0
            break;
254
255
0
        case SearchAlgorithms2::APPROXIMATE:
256
0
            fnForward = &TextSearch::ApproxSrchFrwrd;
257
0
            fnBackward = &TextSearch::ApproxSrchBkwrd;
258
259
0
            pWLD.reset( new WLevDistance( sSrchStr.getStr(), aSrchPara.changedChars,
260
0
                    aSrchPara.insertedChars, aSrchPara.deletedChars,
261
0
                    0 != (SearchFlags::LEV_RELAXED & aSrchPara.searchFlag ) ) );
262
263
0
            nLimit = pWLD->GetLimit();
264
0
            break;
265
266
2
        case SearchAlgorithms2::WILDCARD:
267
2
            mcWildcardEscapeChar = static_cast<sal_uInt32>(aSrchPara.WildcardEscapeCharacter);
268
2
            mbWildcardAllowSubstring = ((aSrchPara.searchFlag & SearchFlags::WILD_MATCH_SELECTION) == 0);
269
2
            fnForward = &TextSearch::WildcardSrchFrwrd;
270
2
            fnBackward = &TextSearch::WildcardSrchBkwrd;
271
2
            break;
272
273
0
        default:
274
0
            SAL_WARN("i18npool","TextSearch::setOptions2 - default what?");
275
0
            [[fallthrough]];
276
4
        case SearchAlgorithms2::ABSOLUTE:
277
4
            fnForward = &TextSearch::NSrchFrwrd;
278
4
            fnBackward = &TextSearch::NSrchBkwrd;
279
4
            break;
280
6
    }
281
6
}
282
283
void TextSearch::setOptions( const SearchOptions& rOptions )
284
3
{
285
3
    sal_Int16 nAlgorithmType2;
286
3
    switch (rOptions.algorithmType)
287
3
    {
288
0
        case SearchAlgorithms_REGEXP:
289
0
            nAlgorithmType2 = SearchAlgorithms2::REGEXP;
290
0
            break;
291
0
        case SearchAlgorithms_APPROXIMATE:
292
0
            nAlgorithmType2 = SearchAlgorithms2::APPROXIMATE;
293
0
            break;
294
0
        default:
295
0
            SAL_WARN("i18npool","TextSearch::setOptions - default what?");
296
0
            [[fallthrough]];
297
3
        case SearchAlgorithms_ABSOLUTE:
298
3
            nAlgorithmType2 = SearchAlgorithms2::ABSOLUTE;
299
3
            break;
300
3
    }
301
    // It would be nice if an inherited struct had a ctor that takes an
302
    // instance of the object the struct derived from...
303
3
    SearchOptions2 aOptions2(
304
3
            rOptions.algorithmType,
305
3
            rOptions.searchFlag,
306
3
            rOptions.searchString,
307
3
            rOptions.replaceString,
308
3
            rOptions.Locale,
309
3
            rOptions.changedChars,
310
3
            rOptions.deletedChars,
311
3
            rOptions.insertedChars,
312
3
            rOptions.transliterateFlags,
313
3
            nAlgorithmType2,
314
3
            0   // no wildcard search, no escape character...
315
3
            );
316
3
    setOptions2( aOptions2);
317
3
}
318
319
static sal_Int32 FindPosInSeq_Impl( const Sequence <sal_Int32>& rOff, sal_Int32 nPos )
320
0
{
321
0
    auto pOff = std::find_if(rOff.begin(), rOff.end(),
322
0
        [nPos](const sal_Int32 nOff) { return nOff >= nPos; });
323
0
    return static_cast<sal_Int32>(std::distance(rOff.begin(), pOff));
324
0
}
325
326
static auto nextCodePoint(std::u16string_view rStr, sal_Int32 nPos)
327
181
{
328
181
    o3tl::iterateCodePoints(rStr, &nPos);
329
181
    return nPos;
330
181
}
331
332
SearchResult TextSearch::searchForward( const OUString& searchStr, sal_Int32 startPos, sal_Int32 endPos )
333
1.31k
{
334
1.31k
    std::unique_lock g(m_aMutex);
335
336
1.31k
    SearchResult sres;
337
338
1.31k
    OUString in_str(searchStr);
339
340
    // in non-regex mode, allow searching typographical apostrophe with the ASCII one
341
    // to avoid regression after using automatic conversion to U+2019 during typing in Writer
342
1.31k
    bool bReplaceApostrophe = bSearchApostrophe && isReplacePunctuation(in_str);
343
344
1.31k
    bUsePrimarySrchStr = true;
345
346
1.31k
    if ( xTranslit.is() )
347
1.31k
    {
348
        // apply normal transliteration (1<->1, 1<->0)
349
350
1.31k
        sal_Int32 nInStartPos = startPos;
351
1.31k
        if (pRegexMatcher && startPos > 0)
352
0
        {
353
            // tdf#89665, tdf#75806: An optimization to avoid transliterating the whole string, yet
354
            // transliterate enough of the leading text to allow sensible look-behind assertions.
355
            // 100 is chosen arbitrarily in the hope that look-behind assertions would largely fit.
356
            // See http://userguide.icu-project.org/strings/regexp for look-behind assertion syntax.
357
            // When search regex doesn't start with an assertion, 3 is to allow startPos to be in
358
            // the middle of a surrogate pair, preceded by another surrogate pair.
359
0
            const sal_Int32 nMaxLeadingLen = aSrchPara.searchString.startsWith("(?") ? 100 : 3;
360
0
            nInStartPos -= std::min(nMaxLeadingLen, startPos);
361
0
        }
362
1.31k
        sal_Int32 nInEndPos = endPos;
363
1.31k
        if (pRegexMatcher && endPos < searchStr.getLength())
364
0
        {
365
            // tdf#65038: ditto for look-ahead assertions
366
0
            const sal_Int32 nMaxTrailingLen = aSrchPara.searchString.endsWith(")") ? 100 : 3;
367
0
            nInEndPos += std::min(nMaxTrailingLen, searchStr.getLength() - endPos);
368
0
        }
369
370
1.31k
        css::uno::Sequence<sal_Int32> offset(nInEndPos - nInStartPos);
371
1.31k
        in_str = xTranslit->transliterate(searchStr, nInStartPos, nInEndPos - nInStartPos, offset);
372
373
1.31k
        if ( bReplaceApostrophe )
374
0
            in_str = replacePunctuation(in_str);
375
376
        // JP 20.6.2001: also the start and end positions must be corrected!
377
1.31k
        sal_Int32 newStartPos =
378
1.31k
            (startPos == 0) ? 0 : FindPosInSeq_Impl( offset, startPos );
379
380
1.31k
        sal_Int32 newEndPos = (endPos < searchStr.getLength())
381
1.31k
            ? FindPosInSeq_Impl( offset, endPos )
382
1.31k
            : in_str.getLength();
383
384
1.31k
        sres = (this->*fnForward)( g, in_str, newStartPos, newEndPos );
385
386
        // Map offsets back to untransliterated string.
387
1.31k
        const sal_Int32 nOffsets = offset.getLength();
388
1.31k
        if (nOffsets)
389
1.31k
        {
390
1.31k
            auto sres_startOffsetRange = asNonConstRange(sres.startOffset);
391
1.31k
            auto sres_endOffsetRange = asNonConstRange(sres.endOffset);
392
            // For regex nGroups is the number of groups+1 with group 0 being
393
            // the entire match.
394
1.31k
            const sal_Int32 nGroups = sres.startOffset.getLength();
395
1.49k
            for ( sal_Int32 k = 0; k < nGroups; k++ )
396
181
            {
397
181
                const sal_Int32 nStart = sres.startOffset[k];
398
                // Result offsets are negative (-1) if a group expression was
399
                // not matched.
400
181
                if (nStart >= 0)
401
181
                    sres_startOffsetRange[k] = (nStart < nOffsets ? offset[nStart]
402
181
                            : nextCodePoint(searchStr, offset[nOffsets - 1]));
403
                // JP 20.6.2001: end is ever exclusive and then don't return
404
                //               the position of the next character - return the
405
                //               next position behind the last found character!
406
                //               "a b c" find "b" must return 2,3 and not 2,4!!!
407
181
                const sal_Int32 nStop = sres.endOffset[k];
408
181
                if (nStop >= 0)
409
181
                {
410
181
                    if (nStop > 0)
411
181
                        sres_endOffsetRange[k] = nextCodePoint(searchStr,
412
181
                                offset[(nStop <= nOffsets ? nStop : nOffsets) - 1]);
413
0
                    else
414
0
                        sres_endOffsetRange[k] = offset[0];
415
181
                }
416
181
            }
417
1.31k
        }
418
1.31k
    }
419
0
    else
420
0
    {
421
0
        if ( bReplaceApostrophe )
422
0
            in_str = in_str.replace(u'\u2019', '\'');
423
424
0
        sres = (this->*fnForward)( g, in_str, startPos, endPos );
425
0
    }
426
427
1.31k
    if ( xTranslit2.is() && aSrchPara.AlgorithmType2 != SearchAlgorithms2::REGEXP)
428
0
    {
429
0
        SearchResult sres2;
430
431
0
        in_str = searchStr;
432
0
        css::uno::Sequence <sal_Int32> offset( in_str.getLength());
433
434
0
        in_str = xTranslit2->transliterate( searchStr, 0, in_str.getLength(), offset );
435
436
0
        if( startPos )
437
0
            startPos = FindPosInSeq_Impl( offset, startPos );
438
439
0
        if( endPos < searchStr.getLength() )
440
0
            endPos = FindPosInSeq_Impl( offset, endPos );
441
0
        else
442
0
            endPos = in_str.getLength();
443
444
0
        bUsePrimarySrchStr = false;
445
0
        sres2 = (this->*fnForward)( g, in_str, startPos, endPos );
446
0
        auto sres2_startOffsetRange = asNonConstRange(sres2.startOffset);
447
0
        auto sres2_endOffsetRange = asNonConstRange(sres2.endOffset);
448
449
0
        for ( int k = 0; k < sres2.startOffset.getLength(); k++ )
450
0
        {
451
0
            if (sres2.startOffset[k])
452
0
                sres2_startOffsetRange[k] = nextCodePoint(searchStr, offset[sres2.startOffset[k]-1]);
453
0
            if (sres2.endOffset[k])
454
0
                sres2_endOffsetRange[k] = nextCodePoint(searchStr, offset[sres2.endOffset[k]-1]);
455
0
        }
456
457
        // pick first and long one
458
0
        if ( sres.subRegExpressions == 0)
459
0
            return sres2;
460
0
        if ( sres2.subRegExpressions == 1)
461
0
        {
462
0
            if ( sres.startOffset[0] > sres2.startOffset[0])
463
0
                return sres2;
464
0
            else if ( sres.startOffset[0] == sres2.startOffset[0] &&
465
0
                    sres.endOffset[0] < sres2.endOffset[0])
466
0
                return sres2;
467
0
        }
468
0
    }
469
470
1.31k
    return sres;
471
1.31k
}
472
473
SearchResult TextSearch::searchBackward( const OUString& searchStr, sal_Int32 startPos, sal_Int32 endPos )
474
0
{
475
0
    std::unique_lock g(m_aMutex);
476
477
0
    SearchResult sres;
478
479
0
    OUString in_str(searchStr);
480
481
    // in non-regex mode, allow searching typographical apostrophe with the ASCII one
482
    // to avoid regression after using automatic conversion to U+2019 during typing in Writer
483
0
    bool bReplaceApostrophe = bSearchApostrophe && isReplacePunctuation(in_str);
484
485
0
    bUsePrimarySrchStr = true;
486
487
0
    if ( xTranslit.is() )
488
0
    {
489
        // apply only simple 1<->1 transliteration here
490
0
        css::uno::Sequence<sal_Int32> offset(startPos - endPos);
491
0
        in_str = xTranslit->transliterate( searchStr, endPos, startPos - endPos, offset );
492
493
0
        if ( bReplaceApostrophe )
494
0
            in_str = replacePunctuation(in_str);
495
496
        // JP 20.6.2001: also the start and end positions must be corrected!
497
0
        sal_Int32 const newStartPos = (startPos < searchStr.getLength())
498
0
            ? FindPosInSeq_Impl( offset, startPos )
499
0
            : in_str.getLength();
500
501
0
        sal_Int32 const newEndPos =
502
0
            (endPos == 0) ? 0 : FindPosInSeq_Impl( offset, endPos );
503
504
        // TODO: this would need nExtraOffset handling to avoid $ matching
505
        // if (pRegexMatcher && startPos < searchStr.getLength())
506
        // but that appears to be impossible with ICU regex
507
508
0
        sres = (this->*fnBackward)( g, in_str, newStartPos, newEndPos );
509
510
        // Map offsets back to untransliterated string.
511
0
        const sal_Int32 nOffsets = offset.getLength();
512
0
        if (nOffsets)
513
0
        {
514
0
            auto sres_startOffsetRange = asNonConstRange(sres.startOffset);
515
0
            auto sres_endOffsetRange = asNonConstRange(sres.endOffset);
516
            // For regex nGroups is the number of groups+1 with group 0 being
517
            // the entire match.
518
0
            const sal_Int32 nGroups = sres.startOffset.getLength();
519
0
            for ( sal_Int32 k = 0; k < nGroups; k++ )
520
0
            {
521
0
                const sal_Int32 nStart = sres.startOffset[k];
522
                // Result offsets are negative (-1) if a group expression was
523
                // not matched.
524
0
                if (nStart >= 0)
525
0
                {
526
0
                    if (nStart > 0)
527
0
                        sres_startOffsetRange[k] = nextCodePoint(searchStr,
528
0
                                offset[(nStart <= nOffsets ? nStart : nOffsets) - 1]);
529
0
                    else
530
0
                        sres_startOffsetRange[k] = offset[0];
531
0
                }
532
                // JP 20.6.2001: end is ever exclusive and then don't return
533
                //               the position of the next character - return the
534
                //               next position behind the last found character!
535
                //               "a b c" find "b" must return 2,3 and not 2,4!!!
536
0
                const sal_Int32 nStop = sres.endOffset[k];
537
0
                if (nStop >= 0)
538
0
                    sres_endOffsetRange[k] = (nStop < nOffsets ? offset[nStop]
539
0
                            : nextCodePoint(searchStr, offset[nOffsets - 1]));
540
0
            }
541
0
        }
542
0
    }
543
0
    else
544
0
    {
545
0
        if ( bReplaceApostrophe )
546
0
            in_str = replacePunctuation(in_str);
547
548
0
        sres = (this->*fnBackward)( g, in_str, startPos, endPos );
549
0
    }
550
551
0
    if ( xTranslit2.is() && aSrchPara.AlgorithmType2 != SearchAlgorithms2::REGEXP )
552
0
    {
553
0
        SearchResult sres2;
554
555
0
        in_str = searchStr;
556
0
        css::uno::Sequence <sal_Int32> offset( in_str.getLength());
557
558
0
        in_str = xTranslit2->transliterate(searchStr, 0, in_str.getLength(), offset);
559
560
0
        if( startPos < searchStr.getLength() )
561
0
            startPos = FindPosInSeq_Impl( offset, startPos );
562
0
        else
563
0
            startPos = in_str.getLength();
564
565
0
        if( endPos )
566
0
            endPos = FindPosInSeq_Impl( offset, endPos );
567
568
0
        bUsePrimarySrchStr = false;
569
0
        sres2 = (this->*fnBackward)( g, in_str, startPos, endPos );
570
0
        auto sres2_startOffsetRange = asNonConstRange(sres2.startOffset);
571
0
        auto sres2_endOffsetRange = asNonConstRange(sres2.endOffset);
572
573
0
        for( int k = 0; k < sres2.startOffset.getLength(); k++ )
574
0
        {
575
0
            if (sres2.startOffset[k])
576
0
                sres2_startOffsetRange[k] = nextCodePoint(searchStr, offset[sres2.startOffset[k]-1]);
577
0
            if (sres2.endOffset[k])
578
0
                sres2_endOffsetRange[k] = nextCodePoint(searchStr, offset[sres2.endOffset[k]-1]);
579
0
        }
580
581
        // pick last and long one
582
0
        if ( sres.subRegExpressions == 0 )
583
0
            return sres2;
584
0
        if ( sres2.subRegExpressions == 1 )
585
0
        {
586
0
            if ( sres.startOffset[0] < sres2.startOffset[0] )
587
0
                return sres2;
588
0
            if ( sres.startOffset[0] == sres2.startOffset[0] &&
589
0
                    sres.endOffset[0] > sres2.endOffset[0] )
590
0
                return sres2;
591
0
        }
592
0
    }
593
594
0
    return sres;
595
0
}
596
597
598
bool TextSearch::IsDelimiter( const OUString& rStr, sal_Int32 nPos ) const
599
0
{
600
0
    bool bRet = true;
601
0
    if( '\x7f' != rStr[nPos])
602
0
    {
603
0
        if ( !xCharClass.is() )
604
0
             xCharClass = CharacterClassification::create( m_xContext );
605
0
        sal_Int32 nCType = xCharClass->getCharacterType( rStr, nPos,
606
0
                aSrchPara.Locale );
607
0
        if( 0 != (( KCharacterType::DIGIT | KCharacterType::ALPHA |
608
0
                        KCharacterType::LETTER ) & nCType ) )
609
0
            bRet = false;
610
0
    }
611
0
    return bRet;
612
0
}
613
614
// --------- helper methods for Boyer-Moore like text searching ----------
615
// TODO: use ICU's regex UREGEX_LITERAL mode instead when it becomes available
616
617
void TextSearch::MakeForwardTab()
618
24
{
619
    // create the jumptable for the search text
620
621
24
    if( pJumpTable && bIsForwardTab )
622
23
    {
623
23
        return; // the jumpTable is ok
624
23
    }
625
1
    bIsForwardTab = true;
626
627
1
    sal_Int32 n, nLen = sSrchStr.getLength();
628
1
    pJumpTable.reset( new TextSearchJumpTable );
629
630
9
    for( n = 0; n < nLen - 1; ++n )
631
8
    {
632
8
        sal_Unicode cCh = sSrchStr[n];
633
8
        sal_Int32 nDiff = nLen - n - 1;
634
8
        TextSearchJumpTable::value_type aEntry( cCh, nDiff );
635
636
8
        ::std::pair< TextSearchJumpTable::iterator, bool > aPair =
637
8
            pJumpTable->insert( aEntry );
638
8
        if ( !aPair.second )
639
1
            (*(aPair.first)).second = nDiff;
640
8
    }
641
1
}
642
643
void TextSearch::MakeForwardTab2()
644
0
{
645
    // create the jumptable for the search text
646
0
    if( pJumpTable2 && bIsForwardTab )
647
0
    {
648
0
        return;        // the jumpTable is ok
649
0
    }
650
0
    bIsForwardTab = true;
651
652
0
    sal_Int32 n, nLen = sSrchStr2.getLength();
653
0
    pJumpTable2.reset( new TextSearchJumpTable );
654
655
0
    for( n = 0; n < nLen - 1; ++n )
656
0
    {
657
0
        sal_Unicode cCh = sSrchStr2[n];
658
0
        sal_Int32 nDiff = nLen - n - 1;
659
660
0
        TextSearchJumpTable::value_type aEntry( cCh, nDiff );
661
0
        ::std::pair< TextSearchJumpTable::iterator, bool > aPair =
662
0
            pJumpTable2->insert( aEntry );
663
0
        if ( !aPair.second )
664
0
            (*(aPair.first)).second = nDiff;
665
0
    }
666
0
}
667
668
void TextSearch::MakeBackwardTab()
669
0
{
670
    // create the jumptable for the search text
671
0
    if( pJumpTable && !bIsForwardTab)
672
0
    {
673
0
        return;   // the jumpTable is ok
674
0
    }
675
0
    bIsForwardTab = false;
676
677
0
    sal_Int32 n, nLen = sSrchStr.getLength();
678
0
    pJumpTable.reset( new TextSearchJumpTable );
679
680
0
    for( n = nLen-1; n > 0; --n )
681
0
    {
682
0
        sal_Unicode cCh = sSrchStr[n];
683
0
        TextSearchJumpTable::value_type aEntry( cCh, n );
684
0
        ::std::pair< TextSearchJumpTable::iterator, bool > aPair =
685
0
            pJumpTable->insert( aEntry );
686
0
        if ( !aPair.second )
687
0
            (*(aPair.first)).second = n;
688
0
    }
689
0
}
690
691
void TextSearch::MakeBackwardTab2()
692
0
{
693
    // create the jumptable for the search text
694
0
    if( pJumpTable2 && !bIsForwardTab )
695
0
    {
696
0
        return;    // the jumpTable is ok
697
0
    }
698
0
    bIsForwardTab = false;
699
700
0
    sal_Int32 n, nLen = sSrchStr2.getLength();
701
0
    pJumpTable2.reset( new TextSearchJumpTable );
702
703
0
    for( n = nLen-1; n > 0; --n )
704
0
    {
705
0
        sal_Unicode cCh = sSrchStr2[n];
706
0
        TextSearchJumpTable::value_type aEntry( cCh, n );
707
0
        ::std::pair< TextSearchJumpTable::iterator, bool > aPair =
708
0
            pJumpTable2->insert( aEntry );
709
0
        if ( !aPair.second )
710
0
            (*(aPair.first)).second = n;
711
0
    }
712
0
}
713
714
sal_Int32 TextSearch::GetDiff( const sal_Unicode cChr ) const
715
165
{
716
165
    TextSearchJumpTable *pJump;
717
165
    OUString sSearchKey;
718
719
165
    if ( bUsePrimarySrchStr ) {
720
165
        pJump = pJumpTable.get();
721
165
        sSearchKey = sSrchStr;
722
165
    } else {
723
0
        pJump = pJumpTable2.get();
724
0
        sSearchKey = sSrchStr2;
725
0
    }
726
727
165
    TextSearchJumpTable::const_iterator iLook = pJump->find( cChr );
728
165
    if ( iLook == pJump->end() )
729
120
        return sSearchKey.getLength();
730
45
    return (*iLook).second;
731
165
}
732
733
734
SearchResult TextSearch::NSrchFrwrd( std::unique_lock<std::mutex>& /*rGuard*/, const OUString& searchStr, sal_Int32 startPos, sal_Int32 endPos )
735
24
{
736
24
    SearchResult aRet;
737
24
    aRet.subRegExpressions = 0;
738
739
24
    OUString sSearchKey = bUsePrimarySrchStr ? sSrchStr : sSrchStr2;
740
741
24
    sal_Int32 nSuchIdx = searchStr.getLength();
742
24
    sal_Int32 nEnd = endPos;
743
24
    if( !nSuchIdx || !sSearchKey.getLength() || sSearchKey.getLength() > nSuchIdx )
744
0
        return aRet;
745
746
747
24
    if( nEnd < sSearchKey.getLength() )   // position inside the search region ?
748
0
        return aRet;
749
750
24
    nEnd -= sSearchKey.getLength();
751
752
24
    if (bUsePrimarySrchStr)
753
24
      MakeForwardTab();                   // create the jumptable
754
0
    else
755
0
      MakeForwardTab2();
756
757
24
    for (sal_Int32 nCmpIdx = startPos; // start position for the search
758
189
            nCmpIdx <= nEnd;
759
165
            nCmpIdx += GetDiff( searchStr[nCmpIdx + sSearchKey.getLength()-1]))
760
180
    {
761
180
        nSuchIdx = sSearchKey.getLength() - 1;
762
315
        while( nSuchIdx >= 0 && sSearchKey[nSuchIdx] == searchStr[nCmpIdx + nSuchIdx])
763
150
        {
764
150
            if( nSuchIdx == 0 )
765
15
            {
766
15
                if( SearchFlags::NORM_WORD_ONLY & aSrchPara.searchFlag )
767
0
                {
768
0
                    sal_Int32 nFndEnd = nCmpIdx + sSearchKey.getLength();
769
0
                    bool bAtStart = !nCmpIdx;
770
0
                    bool bAtEnd = nFndEnd == endPos;
771
0
                    bool bDelimBefore = bAtStart || IsDelimiter( searchStr, nCmpIdx-1 );
772
0
                    bool bDelimBehind = bAtEnd || IsDelimiter(  searchStr, nFndEnd );
773
                    //  *       1 -> only one word in the paragraph
774
                    //  *       2 -> at begin of paragraph
775
                    //  *       3 -> at end of paragraph
776
                    //  *       4 -> inside the paragraph
777
0
                    if( !(  ( bAtStart && bAtEnd ) ||           // 1
778
0
                                ( bAtStart && bDelimBehind ) ||     // 2
779
0
                                ( bAtEnd && bDelimBefore ) ||       // 3
780
0
                                ( bDelimBefore && bDelimBehind )))  // 4
781
0
                        break;
782
0
                }
783
784
15
                aRet.subRegExpressions = 1;
785
15
                aRet.startOffset = { nCmpIdx };
786
15
                aRet.endOffset = { nCmpIdx + sSearchKey.getLength() };
787
788
15
                return aRet;
789
15
            }
790
135
            else
791
135
                nSuchIdx--;
792
150
        }
793
180
    }
794
9
    return aRet;
795
24
}
796
797
SearchResult TextSearch::NSrchBkwrd( std::unique_lock<std::mutex>& /*rGuard*/, const OUString& searchStr, sal_Int32 startPos, sal_Int32 endPos )
798
0
{
799
0
    SearchResult aRet;
800
0
    aRet.subRegExpressions = 0;
801
802
0
    OUString sSearchKey = bUsePrimarySrchStr ? sSrchStr : sSrchStr2;
803
804
0
    sal_Int32 nSuchIdx = searchStr.getLength();
805
0
    sal_Int32 nEnd = endPos;
806
0
    if( nSuchIdx == 0 || sSearchKey.isEmpty() || sSearchKey.getLength() > nSuchIdx)
807
0
        return aRet;
808
809
0
    if (bUsePrimarySrchStr)
810
0
        MakeBackwardTab();                  // create the jumptable
811
0
    else
812
0
        MakeBackwardTab2();
813
814
0
    if( nEnd == nSuchIdx )                  // end position for the search
815
0
        nEnd = sSearchKey.getLength();
816
0
    else
817
0
        nEnd += sSearchKey.getLength();
818
819
0
    sal_Int32 nCmpIdx = startPos;          // start position for the search
820
821
0
    while (nCmpIdx >= nEnd)
822
0
    {
823
0
        nSuchIdx = 0;
824
0
        while( nSuchIdx < sSearchKey.getLength() && sSearchKey[nSuchIdx] ==
825
0
                searchStr[nCmpIdx + nSuchIdx - sSearchKey.getLength()] )
826
0
            nSuchIdx++;
827
0
        if( nSuchIdx >= sSearchKey.getLength() )
828
0
        {
829
0
            if( SearchFlags::NORM_WORD_ONLY & aSrchPara.searchFlag )
830
0
            {
831
0
                sal_Int32 nFndStt = nCmpIdx - sSearchKey.getLength();
832
0
                bool bAtStart = !nFndStt;
833
0
                bool bAtEnd = nCmpIdx == startPos;
834
0
                bool bDelimBehind = bAtEnd || IsDelimiter( searchStr, nCmpIdx );
835
0
                bool bDelimBefore = bAtStart || // begin of paragraph
836
0
                    IsDelimiter( searchStr, nFndStt-1 );
837
                //  *       1 -> only one word in the paragraph
838
                //  *       2 -> at begin of paragraph
839
                //  *       3 -> at end of paragraph
840
                //  *       4 -> inside the paragraph
841
0
                if( ( bAtStart && bAtEnd ) ||           // 1
842
0
                        ( bAtStart && bDelimBehind ) ||     // 2
843
0
                        ( bAtEnd && bDelimBefore ) ||       // 3
844
0
                        ( bDelimBefore && bDelimBehind ))   // 4
845
0
                {
846
0
                    aRet.subRegExpressions = 1;
847
0
                    aRet.startOffset = { nCmpIdx };
848
0
                    aRet.endOffset = { nCmpIdx - sSearchKey.getLength() };
849
0
                    return aRet;
850
0
                }
851
0
            }
852
0
            else
853
0
            {
854
0
                aRet.subRegExpressions = 1;
855
0
                aRet.startOffset = { nCmpIdx };
856
0
                aRet.endOffset = { nCmpIdx - sSearchKey.getLength() };
857
0
                return aRet;
858
0
            }
859
0
        }
860
0
        nSuchIdx = GetDiff( searchStr[nCmpIdx - sSearchKey.getLength()] );
861
0
        if( nCmpIdx < nSuchIdx )
862
0
            return aRet;
863
0
        nCmpIdx -= nSuchIdx;
864
0
    }
865
0
    return aRet;
866
0
}
867
868
void TextSearch::RESrchPrepare( const css::util::SearchOptions2& rOptions)
869
0
{
870
0
    TransliterationFlags transliterateFlags = static_cast<TransliterationFlags>(rOptions.transliterateFlags);
871
    // select the transliterated pattern string
872
0
    const OUString& rPatternStr =
873
0
        (isSimpleTrans(transliterateFlags) ? sSrchStr
874
0
        : (isComplexTrans(transliterateFlags) ? sSrchStr2 : rOptions.searchString));
875
876
0
    sal_uInt32 nIcuSearchFlags = UREGEX_UWORD; // request UAX#29 unicode capability
877
    // map css::util::SearchFlags to ICU uregex.h flags
878
    // TODO: REG_EXTENDED, REG_NOT_BEGINOFLINE, REG_NOT_ENDOFLINE
879
    // REG_NEWLINE is neither properly defined nor used anywhere => not implemented
880
    // REG_NOSUB is not used anywhere => not implemented
881
    // NORM_WORD_ONLY is only used for SearchAlgorithm==Absolute
882
    // LEV_RELAXED is only used for SearchAlgorithm==Approximate
883
    // Note that the search flag ALL_IGNORE_CASE is deprecated in UNO
884
    // probably because the transliteration flag IGNORE_CASE handles it as well.
885
0
    if( (rOptions.searchFlag & css::util::SearchFlags::ALL_IGNORE_CASE) != 0
886
0
    ||  (transliterateFlags & TransliterationFlags::IGNORE_CASE))
887
0
        nIcuSearchFlags |= UREGEX_CASE_INSENSITIVE;
888
0
    UErrorCode nIcuErr = U_ZERO_ERROR;
889
    // assumption: transliteration didn't mangle regexp control chars
890
0
    icu::UnicodeString aIcuSearchPatStr( reinterpret_cast<const UChar*>(rPatternStr.getStr()), rPatternStr.getLength());
891
0
#ifndef DISABLE_WORDBOUND_EMULATION
892
    // for convenience specific syntax elements of the old regex engine are emulated
893
    // - by replacing \< with "word-break followed by a look-ahead word-char"
894
0
    static const icu::UnicodeString aChevronPatternB( "\\\\<", -1, icu::UnicodeString::kInvariant);
895
0
    static const icu::UnicodeString aChevronReplaceB( "\\\\b(?=\\\\w)", -1, icu::UnicodeString::kInvariant);
896
0
    static icu::RegexMatcher aChevronMatcherB( aChevronPatternB, 0, nIcuErr);
897
0
    aChevronMatcherB.reset( aIcuSearchPatStr);
898
0
    aIcuSearchPatStr = aChevronMatcherB.replaceAll( aChevronReplaceB, nIcuErr);
899
0
    aChevronMatcherB.reset();
900
    // - by replacing \> with "look-behind word-char followed by a word-break"
901
0
    static const icu::UnicodeString aChevronPatternE( "\\\\>", -1, icu::UnicodeString::kInvariant);
902
0
    static const icu::UnicodeString aChevronReplaceE( "(?<=\\\\w)\\\\b", -1, icu::UnicodeString::kInvariant);
903
0
    static icu::RegexMatcher aChevronMatcherE( aChevronPatternE, 0, nIcuErr);
904
0
    aChevronMatcherE.reset( aIcuSearchPatStr);
905
0
    aIcuSearchPatStr = aChevronMatcherE.replaceAll( aChevronReplaceE, nIcuErr);
906
0
    aChevronMatcherE.reset();
907
0
#endif
908
0
    pRegexMatcher.reset( new icu::RegexMatcher( aIcuSearchPatStr, nIcuSearchFlags, nIcuErr) );
909
0
    if (nIcuErr)
910
0
    {
911
0
        SAL_INFO( "i18npool", "TextSearch::RESrchPrepare UErrorCode " << nIcuErr);
912
0
        pRegexMatcher.reset();
913
0
    }
914
0
    else
915
0
    {
916
        // Pathological patterns may result in exponential run time making the
917
        // application appear to be frozen. Limit that. Documentation for this
918
        // call says
919
        // https://unicode-org.github.io/icu-docs/apidoc/released/icu4c/classicu_1_1RegexMatcher.html#a6ebcfcab4fe6a38678c0291643a03a00
920
        // "The units of the limit are steps of the match engine.
921
        // Correspondence with actual processor time will depend on the speed
922
        // of the processor and the details of the specific pattern, but will
923
        // typically be on the order of milliseconds."
924
        // Just what is a good value? 42 is always an answer ... the 23 enigma
925
        // as well... which on the dev's machine is roughly 50 seconds with the
926
        // pattern of fdo#70627.
927
        /* TODO: make this a configuration settable value and possibly take
928
         * complexity of expression into account and maybe even length of text
929
         * to be matched; currently (2013-11-25) that is at most one 64k
930
         * paragraph per RESrchFrwrd()/RESrchBkwrd() call. */
931
0
        pRegexMatcher->setTimeLimit( 23*1000, nIcuErr);
932
0
    }
933
0
}
934
935
936
static bool lcl_findRegex(std::unique_ptr<icu::RegexMatcher> const& pRegexMatcher,
937
                          sal_Int32 nStartPos, sal_Int32 nEndPos, UErrorCode& rIcuErr)
938
0
{
939
0
    pRegexMatcher->region(nStartPos, nEndPos, rIcuErr);
940
0
    pRegexMatcher->useAnchoringBounds(false); // use whole text's anchoring bounds, not region's
941
0
    pRegexMatcher->useTransparentBounds(true); // take text outside of the region into account for
942
                                               // look-ahead/behind assertions
943
944
0
    if (!pRegexMatcher->find(rIcuErr))
945
0
    {
946
        /* TODO: future versions could pass the UErrorCode or translations
947
         * thereof to the caller, for example to inform the user of
948
         * U_REGEX_TIME_OUT. The strange thing though is that an error is set
949
         * only after the second call that returns immediately and not if
950
         * timeout occurred on the first call?!? */
951
0
        SAL_INFO( "i18npool", "lcl_findRegex UErrorCode " << rIcuErr);
952
0
        return false;
953
0
    }
954
0
    return true;
955
0
}
956
957
SearchResult TextSearch::RESrchFrwrd( std::unique_lock<std::mutex>& /*rGuard*/, const OUString& searchStr,
958
                                      sal_Int32 startPos, sal_Int32 endPos )
959
0
{
960
0
    SearchResult aRet;
961
0
    aRet.subRegExpressions = 0;
962
0
    if( !pRegexMatcher)
963
0
        return aRet;
964
965
0
    if( endPos > searchStr.getLength())
966
0
        endPos = searchStr.getLength();
967
968
    // use the ICU RegexMatcher to find the matches
969
0
    UErrorCode nIcuErr = U_ZERO_ERROR;
970
0
    const icu::UnicodeString aSearchTargetStr(false, reinterpret_cast<const UChar*>(searchStr.getStr()),
971
0
                                        searchStr.getLength());
972
0
    pRegexMatcher->reset( aSearchTargetStr);
973
0
    if (!lcl_findRegex( pRegexMatcher, startPos, endPos, nIcuErr))
974
0
        return aRet;
975
976
    // extract the result of the search
977
0
    const int nGroupCount = pRegexMatcher->groupCount();
978
0
    aRet.subRegExpressions = nGroupCount + 1;
979
0
    aRet.startOffset.realloc( aRet.subRegExpressions);
980
0
    auto pstartOffset = aRet.startOffset.getArray();
981
0
    aRet.endOffset.realloc( aRet.subRegExpressions);
982
0
    auto pendOffset = aRet.endOffset.getArray();
983
0
    pstartOffset[0] = pRegexMatcher->start( nIcuErr);
984
0
    pendOffset[0]   = pRegexMatcher->end( nIcuErr);
985
0
    for( int i = 1; i <= nGroupCount; ++i) {
986
0
        pstartOffset[i] = pRegexMatcher->start( i, nIcuErr);
987
0
        pendOffset[i]   = pRegexMatcher->end( i, nIcuErr);
988
0
    }
989
990
0
    return aRet;
991
0
}
992
993
SearchResult TextSearch::RESrchBkwrd( std::unique_lock<std::mutex>& /*rGuard*/, const OUString& searchStr,
994
                                      sal_Int32 startPos, sal_Int32 endPos )
995
0
{
996
    // NOTE: for backwards search callers provide startPos/endPos inverted!
997
0
    SearchResult aRet;
998
0
    aRet.subRegExpressions = 0;
999
0
    if( !pRegexMatcher)
1000
0
        return aRet;
1001
1002
0
    if( startPos > searchStr.getLength())
1003
0
        startPos = searchStr.getLength();
1004
1005
    // use the ICU RegexMatcher to find the matches
1006
    // TODO: use ICU's backward searching once it becomes available
1007
    //       as its replacement using forward search is not as good as the real thing
1008
0
    UErrorCode nIcuErr = U_ZERO_ERROR;
1009
0
    const icu::UnicodeString aSearchTargetStr(false, reinterpret_cast<const UChar*>(searchStr.getStr()),
1010
0
                                        searchStr.getLength());
1011
0
    pRegexMatcher->reset( aSearchTargetStr);
1012
0
    if (!lcl_findRegex( pRegexMatcher, endPos, startPos, nIcuErr))
1013
0
        return aRet;
1014
1015
    // find the last match
1016
0
    int nLastPos = 0;
1017
0
    int nFoundEnd = 0;
1018
0
    int nGoodPos = 0, nGoodEnd = 0;
1019
0
    bool bFirst = true;
1020
0
    do {
1021
0
        nLastPos = pRegexMatcher->start( nIcuErr);
1022
0
        nFoundEnd = pRegexMatcher->end( nIcuErr);
1023
0
        if (nLastPos < nFoundEnd)
1024
0
        {
1025
            // remember last non-zero-length match
1026
0
            nGoodPos = nLastPos;
1027
0
            nGoodEnd = nFoundEnd;
1028
0
        }
1029
0
        if( nFoundEnd >= startPos)
1030
0
            break;
1031
0
        bFirst = false;
1032
0
        if( nFoundEnd == nLastPos)
1033
0
            ++nFoundEnd;
1034
0
    } while( lcl_findRegex( pRegexMatcher, nFoundEnd, startPos, nIcuErr));
1035
1036
    // Ignore all zero-length matches except "$" anchor on first match.
1037
0
    if (nGoodPos == nGoodEnd)
1038
0
    {
1039
0
        if (bFirst && nLastPos == startPos)
1040
0
            nGoodPos = nLastPos;
1041
0
        else
1042
0
            return aRet;
1043
0
    }
1044
1045
    // find last match again to get its details
1046
0
    lcl_findRegex( pRegexMatcher, nGoodPos, startPos, nIcuErr);
1047
1048
    // fill in the details of the last match
1049
0
    const int nGroupCount = pRegexMatcher->groupCount();
1050
0
    aRet.subRegExpressions = nGroupCount + 1;
1051
0
    aRet.startOffset.realloc( aRet.subRegExpressions);
1052
0
    auto pstartOffset = aRet.startOffset.getArray();
1053
0
    aRet.endOffset.realloc( aRet.subRegExpressions);
1054
0
    auto pendOffset = aRet.endOffset.getArray();
1055
    // NOTE: existing users of backward search seem to expect startOfs/endOfs being inverted!
1056
0
    pstartOffset[0] = pRegexMatcher->end( nIcuErr);
1057
0
    pendOffset[0]   = pRegexMatcher->start( nIcuErr);
1058
0
    for( int i = 1; i <= nGroupCount; ++i) {
1059
0
        pstartOffset[i] = pRegexMatcher->end( i, nIcuErr);
1060
0
        pendOffset[i]   = pRegexMatcher->start( i, nIcuErr);
1061
0
    }
1062
1063
0
    return aRet;
1064
0
}
1065
1066
1067
// search for words phonetically
1068
SearchResult TextSearch::ApproxSrchFrwrd( std::unique_lock<std::mutex>& /*rGuard*/, const OUString& searchStr,
1069
                                          sal_Int32 startPos, sal_Int32 endPos )
1070
0
{
1071
0
    SearchResult aRet;
1072
0
    aRet.subRegExpressions = 0;
1073
1074
0
    if( !xBreak.is() )
1075
0
        return aRet;
1076
1077
0
    sal_Int32 nStt, nEnd;
1078
1079
0
    Boundary aWBnd = xBreak->getWordBoundary( searchStr, startPos,
1080
0
            aSrchPara.Locale,
1081
0
            WordType::ANYWORD_IGNOREWHITESPACES, true );
1082
1083
0
    do
1084
0
    {
1085
0
        if( aWBnd.startPos >= endPos )
1086
0
            break;
1087
0
        nStt = aWBnd.startPos < startPos ? startPos : aWBnd.startPos;
1088
0
        nEnd = std::min(aWBnd.endPos, endPos);
1089
1090
0
        if( nStt < nEnd &&
1091
0
                pWLD->WLD( searchStr.getStr() + nStt, nEnd - nStt ) <= nLimit )
1092
0
        {
1093
0
            aRet.subRegExpressions = 1;
1094
0
            aRet.startOffset = { nStt };
1095
0
            aRet.endOffset = { nEnd };
1096
0
            break;
1097
0
        }
1098
1099
0
        nStt = nEnd - 1;
1100
0
        aWBnd = xBreak->nextWord( searchStr, nStt, aSrchPara.Locale,
1101
0
                WordType::ANYWORD_IGNOREWHITESPACES);
1102
0
    } while( aWBnd.startPos != aWBnd.endPos ||
1103
0
            (aWBnd.endPos != searchStr.getLength() && aWBnd.endPos != nEnd) );
1104
    // #i50244# aWBnd.endPos != nEnd : in case there is _no_ word (only
1105
    // whitespace) in searchStr, getWordBoundary() returned startPos,startPos
1106
    // and nextWord() does also => don't loop forever.
1107
0
    return aRet;
1108
0
}
1109
1110
SearchResult TextSearch::ApproxSrchBkwrd( std::unique_lock<std::mutex>& /*rGuard*/, const OUString& searchStr,
1111
                                          sal_Int32 startPos, sal_Int32 endPos )
1112
0
{
1113
0
    SearchResult aRet;
1114
0
    aRet.subRegExpressions = 0;
1115
1116
0
    if( !xBreak.is() )
1117
0
        return aRet;
1118
1119
0
    sal_Int32 nStt, nEnd;
1120
1121
0
    Boundary aWBnd = xBreak->getWordBoundary( searchStr, startPos,
1122
0
            aSrchPara.Locale,
1123
0
            WordType::ANYWORD_IGNOREWHITESPACES, true );
1124
1125
0
    do
1126
0
    {
1127
0
        if( aWBnd.endPos <= endPos )
1128
0
            break;
1129
0
        nStt = aWBnd.startPos < endPos ? endPos : aWBnd.startPos;
1130
0
        nEnd = std::min(aWBnd.endPos, startPos);
1131
1132
0
        if( nStt < nEnd &&
1133
0
                pWLD->WLD( searchStr.getStr() + nStt, nEnd - nStt ) <= nLimit )
1134
0
        {
1135
0
            aRet.subRegExpressions = 1;
1136
0
            aRet.startOffset = { nEnd };
1137
0
            aRet.endOffset = { nStt };
1138
0
            break;
1139
0
        }
1140
0
        if( !nStt )
1141
0
            break;
1142
1143
0
        aWBnd = xBreak->previousWord( searchStr, nStt, aSrchPara.Locale,
1144
0
                WordType::ANYWORD_IGNOREWHITESPACES);
1145
0
    } while( aWBnd.startPos != aWBnd.endPos || aWBnd.endPos != searchStr.getLength() );
1146
0
    return aRet;
1147
0
}
1148
1149
1150
namespace {
1151
void setWildcardMatch( css::util::SearchResult& rRes, sal_Int32 nStartOffset, sal_Int32 nEndOffset )
1152
166
{
1153
166
    rRes.subRegExpressions = 1;
1154
166
    rRes.startOffset = { nStartOffset };
1155
166
    rRes.endOffset = { nEndOffset };
1156
166
}
1157
}
1158
1159
SearchResult TextSearch::WildcardSrchFrwrd( std::unique_lock<std::mutex>& /*rGuard*/, const OUString& searchStr, sal_Int32 nStartPos, sal_Int32 nEndPos )
1160
1.28k
{
1161
1.28k
    SearchResult aRes;
1162
1.28k
    aRes.subRegExpressions = 0;     // no match
1163
1.28k
    sal_Int32 nStartOffset = nStartPos;
1164
1.28k
    sal_Int32 nEndOffset = nEndPos;
1165
1166
1.28k
    const sal_Int32 nStringLen = searchStr.getLength();
1167
1168
    // Forward nStartPos inclusive, nEndPos exclusive, but allow for empty
1169
    // string match with [0,0).
1170
1.28k
    if (nStartPos < 0 || nEndPos > nStringLen || nEndPos < nStartPos ||
1171
1.28k
            (nStartPos == nStringLen && (nStringLen != 0 || nStartPos != nEndPos)))
1172
0
        return aRes;
1173
1174
1.28k
    const OUString& rPattern = (bUsePrimarySrchStr ? sSrchStr : sSrchStr2);
1175
1.28k
    const sal_Int32 nPatternLen = rPattern.getLength();
1176
1177
    // Handle special cases empty pattern and/or string outside of the loop to
1178
    // not add performance penalties there and simplify.
1179
1.28k
    if (nStartPos == nEndPos)
1180
0
    {
1181
0
        sal_Int32 i = 0;
1182
0
        while (i < nPatternLen && rPattern[i] == '*')
1183
0
            ++i;
1184
0
        if (i == nPatternLen)
1185
0
            setWildcardMatch( aRes, nStartOffset, nEndOffset);
1186
0
        return aRes;
1187
0
    }
1188
1189
    // Empty pattern does not match any non-empty string.
1190
1.28k
    if (!nPatternLen)
1191
0
        return aRes;
1192
1193
1.28k
    bool bRewind = false;
1194
1.28k
    sal_uInt32 cPattern = 0;
1195
1.28k
    sal_Int32 nPattern = 0;
1196
1.28k
    sal_Int32 nAfterFakePattern = nPattern;
1197
1.28k
    if (mbWildcardAllowSubstring)
1198
0
    {
1199
        // Fake a leading '*' wildcard.
1200
0
        cPattern = '*';
1201
0
        bRewind = true;
1202
        // Assume a non-'*' pattern character follows. If it is a '*' instead
1203
        // that will be handled in the loop by setting nPat.
1204
0
        sal_uInt32 cu = rPattern.iterateCodePoints( &nAfterFakePattern);
1205
0
        if (cu == mcWildcardEscapeChar && mcWildcardEscapeChar && nAfterFakePattern < nPatternLen)
1206
0
            rPattern.iterateCodePoints( &nAfterFakePattern);
1207
0
    }
1208
1209
1.28k
    sal_Int32 nString = nStartPos, nPat = -1, nStr = -1, nLastAsterisk = -1;
1210
1.28k
    sal_uInt32 cPatternAfterAsterisk = 0;
1211
1.28k
    bool bEscaped = false, bEscapedAfterAsterisk = false;
1212
1213
    // The loop code tries to avoid costly calls to iterateCodePoints() when
1214
    // possible.
1215
1216
1.28k
    do
1217
3.30k
    {
1218
3.30k
        if (bRewind)
1219
0
        {
1220
            // Reuse cPattern after '*', nPattern was correspondingly
1221
            // incremented to point behind cPattern.
1222
0
            bRewind = false;
1223
0
        }
1224
3.30k
        else if (nPattern < nPatternLen)
1225
3.30k
        {
1226
            // nPattern will be incremented by iterateCodePoints().
1227
3.30k
            cPattern = rPattern.iterateCodePoints( &nPattern);
1228
3.30k
            if (cPattern == mcWildcardEscapeChar && mcWildcardEscapeChar && nPattern < nPatternLen)
1229
0
            {
1230
0
                bEscaped = true;
1231
0
                cPattern = rPattern.iterateCodePoints( &nPattern);
1232
0
            }
1233
3.30k
        }
1234
0
        else
1235
0
        {
1236
            // A trailing '*' is handled below.
1237
0
            if (mbWildcardAllowSubstring)
1238
0
            {
1239
                // If the pattern is consumed and substring match allowed we're good.
1240
0
                setWildcardMatch( aRes, nStartOffset, nString);
1241
0
                return aRes;
1242
0
            }
1243
0
            else if (nString < nEndPos && nLastAsterisk >= 0)
1244
0
            {
1245
                // If substring match is not allowed try a greedy '*' match.
1246
0
                nPattern = nLastAsterisk;
1247
0
                continue;   // do
1248
0
            }
1249
0
            else
1250
0
                return aRes;
1251
0
        }
1252
1253
3.30k
        if (cPattern == '*' && !bEscaped)
1254
166
        {
1255
            // '*' is one code unit, so not using iterateCodePoints() is ok.
1256
166
            while (nPattern < nPatternLen && rPattern[nPattern] == '*')
1257
0
                ++nPattern;
1258
1259
166
            if (nPattern >= nPatternLen)
1260
166
            {
1261
                // Last pattern is '*', remaining string matches.
1262
166
                setWildcardMatch( aRes, nStartOffset, nEndOffset);
1263
166
                return aRes;
1264
166
            }
1265
1266
0
            nLastAsterisk = nPattern;   // Remember last encountered '*'.
1267
1268
            // cPattern will be the next non-'*' character, nPattern
1269
            // incremented.
1270
0
            cPattern = rPattern.iterateCodePoints( &nPattern);
1271
0
            if (cPattern == mcWildcardEscapeChar && mcWildcardEscapeChar && nPattern < nPatternLen)
1272
0
            {
1273
0
                bEscaped = true;
1274
0
                cPattern = rPattern.iterateCodePoints( &nPattern);
1275
0
            }
1276
1277
0
            cPatternAfterAsterisk = cPattern;
1278
0
            bEscapedAfterAsterisk = bEscaped;
1279
0
            nPat = nPattern;    // Remember position of pattern behind '*', already incremented.
1280
0
            nStr = nString;     // Remember the current string to be matched.
1281
0
        }
1282
1283
3.14k
        if (nString >= nEndPos)
1284
            // Whatever follows in pattern, string will not match.
1285
0
            return aRes;
1286
1287
        // nString will be incremented by iterateCodePoints().
1288
3.14k
        sal_uInt32 cString = searchStr.iterateCodePoints( &nString);
1289
1290
3.14k
        if ((cPattern != '?' || bEscaped) && cPattern != cString)
1291
1.12k
        {
1292
1.12k
            if (nPat == -1)
1293
                // Non-match already without any '*' pattern.
1294
1.12k
                return aRes;
1295
1296
0
            bRewind = true;
1297
0
            nPattern = nPat;                    // Rewind pattern to character behind '*', already incremented.
1298
0
            cPattern = cPatternAfterAsterisk;
1299
0
            bEscaped = bEscapedAfterAsterisk;
1300
0
            searchStr.iterateCodePoints( &nStr);
1301
0
            nString = nStr;                     // Restore incremented remembered string position.
1302
0
            if (nPat == nAfterFakePattern)
1303
0
            {
1304
                // Next start offset will be the next character.
1305
0
                nStartOffset = nString;
1306
0
            }
1307
0
        }
1308
2.02k
        else
1309
2.02k
        {
1310
            // An unescaped '?' pattern matched any character, or characters
1311
            // matched. Reset only escaped state.
1312
2.02k
            bEscaped = false;
1313
2.02k
        }
1314
3.14k
    }
1315
2.02k
    while (nString < nEndPos);
1316
1317
0
    if (bRewind)
1318
0
        return aRes;
1319
1320
    // Eat trailing '*' pattern that matches anything, including nothing.
1321
    // '*' is one code unit, so not using iterateCodePoints() is ok.
1322
0
    while (nPattern < nPatternLen && rPattern[nPattern] == '*')
1323
0
        ++nPattern;
1324
1325
0
    if (nPattern == nPatternLen)
1326
0
        setWildcardMatch( aRes, nStartOffset, nEndOffset);
1327
0
    return aRes;
1328
0
}
1329
1330
SearchResult TextSearch::WildcardSrchBkwrd( std::unique_lock<std::mutex>& /*rGuard*/, const OUString& searchStr, sal_Int32 nStartPos, sal_Int32 nEndPos )
1331
0
{
1332
0
    SearchResult aRes;
1333
0
    aRes.subRegExpressions = 0;     // no match
1334
1335
0
    sal_Int32 nStartOffset = nStartPos;
1336
0
    sal_Int32 nEndOffset = nEndPos;
1337
1338
0
    const sal_Int32 nStringLen = searchStr.getLength();
1339
1340
    // Backward nStartPos exclusive, nEndPos inclusive, but allow for empty
1341
    // string match with (0,0].
1342
0
    if (nStartPos > nStringLen || nEndPos < 0 || nStartPos < nEndPos ||
1343
0
            (nEndPos == nStringLen && (nStringLen != 0 || nStartPos != nEndPos)))
1344
0
        return aRes;
1345
1346
0
    const OUString& rPattern = (bUsePrimarySrchStr ? sSrchStr : sSrchStr2);
1347
0
    sal_Int32 nPatternLen = rPattern.getLength();
1348
1349
    // Handle special cases empty pattern and/or string outside of the loop to
1350
    // not add performance penalties there and simplify.
1351
0
    if (nStartPos == nEndPos)
1352
0
    {
1353
0
        sal_Int32 i = 0;
1354
0
        while (i < nPatternLen && rPattern[i] == '*')
1355
0
            ++i;
1356
0
        if (i == nPatternLen)
1357
0
            setWildcardMatch( aRes, nStartOffset, nEndOffset);
1358
0
        return aRes;
1359
0
    }
1360
1361
    // Empty pattern does not match any non-empty string.
1362
0
    if (!nPatternLen)
1363
0
        return aRes;
1364
1365
    // Reverse escaped patterns to ease the handling of escapes, keeping escape
1366
    // and following character as one sequence in backward direction.
1367
0
    if ((bUsePrimarySrchStr && maWildcardReversePattern.isEmpty()) ||
1368
0
            (!bUsePrimarySrchStr && maWildcardReversePattern2.isEmpty()))
1369
0
    {
1370
0
        OUStringBuffer aPatternBuf( rPattern);
1371
0
        sal_Int32 nIndex = 0;
1372
0
        while (nIndex < nPatternLen)
1373
0
        {
1374
0
            const sal_Int32 nOld = nIndex;
1375
0
            const sal_uInt32 cu = rPattern.iterateCodePoints( &nIndex);
1376
0
            if (cu == mcWildcardEscapeChar)
1377
0
            {
1378
0
                if (nIndex < nPatternLen)
1379
0
                {
1380
0
                    if (nIndex - nOld == 1)
1381
0
                    {
1382
                        // Simply move code units, we already memorized the one
1383
                        // in 'cu'.
1384
0
                        const sal_Int32 nOld2 = nIndex;
1385
0
                        rPattern.iterateCodePoints( &nIndex);
1386
0
                        for (sal_Int32 i=0; i < nIndex - nOld2; ++i)
1387
0
                            aPatternBuf[nOld+i] = rPattern[nOld2+i];
1388
0
                        aPatternBuf[nIndex-1] = static_cast<sal_Unicode>(cu);
1389
0
                    }
1390
0
                    else
1391
0
                    {
1392
                        // Copy the escape character code units first in the
1393
                        // unlikely case that it would not be of BMP.
1394
0
                        assert(nIndex - nOld == 2);  // it's UTF-16, so...
1395
0
                        sal_Unicode buf[2];
1396
0
                        buf[0] = rPattern[nOld];
1397
0
                        buf[1] = rPattern[nOld+1];
1398
0
                        const sal_Int32 nOld2 = nIndex;
1399
0
                        rPattern.iterateCodePoints( &nIndex);
1400
0
                        for (sal_Int32 i=0; i < nIndex - nOld2; ++i)
1401
0
                            aPatternBuf[nOld+i] = rPattern[nOld2+i];
1402
0
                        aPatternBuf[nIndex-2] = buf[0];
1403
0
                        aPatternBuf[nIndex-1] = buf[1];
1404
0
                    }
1405
0
                }
1406
0
                else
1407
0
                {
1408
                    // Trailing escape would become leading escape, do what?
1409
                    // Eliminate.
1410
0
                    aPatternBuf.remove( nOld, nIndex - nOld);
1411
0
                }
1412
0
            }
1413
0
        }
1414
0
        if (bUsePrimarySrchStr)
1415
0
            maWildcardReversePattern = aPatternBuf.makeStringAndClear();
1416
0
        else
1417
0
            maWildcardReversePattern2 = aPatternBuf.makeStringAndClear();
1418
0
    }
1419
0
    const OUString& rReversePattern = (bUsePrimarySrchStr ? maWildcardReversePattern : maWildcardReversePattern2);
1420
0
    nPatternLen = rReversePattern.getLength();
1421
1422
0
    bool bRewind = false;
1423
0
    sal_uInt32 cPattern = 0;
1424
0
    sal_Int32 nPattern = nPatternLen;
1425
0
    sal_Int32 nAfterFakePattern = nPattern;
1426
0
    if (mbWildcardAllowSubstring)
1427
0
    {
1428
        // Fake a trailing '*' wildcard.
1429
0
        cPattern = '*';
1430
0
        bRewind = true;
1431
        // Assume a non-'*' pattern character follows. If it is a '*' instead
1432
        // that will be handled in the loop by setting nPat.
1433
0
        sal_uInt32 cu = rReversePattern.iterateCodePoints( &nAfterFakePattern, -1);
1434
0
        if (cu == mcWildcardEscapeChar && mcWildcardEscapeChar && nAfterFakePattern > 0)
1435
0
            rReversePattern.iterateCodePoints( &nAfterFakePattern, -1);
1436
0
    }
1437
1438
0
    sal_Int32 nString = nStartPos, nPat = -1, nStr = -1, nLastAsterisk = -1;
1439
0
    sal_uInt32 cPatternAfterAsterisk = 0;
1440
0
    bool bEscaped = false, bEscapedAfterAsterisk = false;
1441
1442
    // The loop code tries to avoid costly calls to iterateCodePoints() when
1443
    // possible.
1444
1445
0
    do
1446
0
    {
1447
0
        if (bRewind)
1448
0
        {
1449
            // Reuse cPattern after '*', nPattern was correspondingly
1450
            // decremented to point before cPattern.
1451
0
            bRewind = false;
1452
0
        }
1453
0
        else if (nPattern > 0)
1454
0
        {
1455
            // nPattern will be decremented by iterateCodePoints().
1456
0
            cPattern = rReversePattern.iterateCodePoints( &nPattern, -1);
1457
0
            if (cPattern == mcWildcardEscapeChar && mcWildcardEscapeChar && nPattern > 0)
1458
0
            {
1459
0
                bEscaped = true;
1460
0
                cPattern = rReversePattern.iterateCodePoints( &nPattern, -1);
1461
0
            }
1462
0
        }
1463
0
        else
1464
0
        {
1465
            // A trailing '*' is handled below.
1466
0
            if (mbWildcardAllowSubstring)
1467
0
            {
1468
                // If the pattern is consumed and substring match allowed we're good.
1469
0
                setWildcardMatch( aRes, nStartOffset, nString);
1470
0
                return aRes;
1471
0
            }
1472
0
            else if (nString > nEndPos && nLastAsterisk >= 0)
1473
0
            {
1474
                // If substring match is not allowed try a greedy '*' match.
1475
0
                nPattern = nLastAsterisk;
1476
0
                continue;   // do
1477
0
            }
1478
0
            else
1479
0
                return aRes;
1480
0
        }
1481
1482
0
        if (cPattern == '*' && !bEscaped)
1483
0
        {
1484
            // '*' is one code unit, so not using iterateCodePoints() is ok.
1485
0
            while (nPattern > 0 && rReversePattern[nPattern-1] == '*')
1486
0
                --nPattern;
1487
1488
0
            if (nPattern <= 0)
1489
0
            {
1490
                // First pattern is '*', remaining string matches.
1491
0
                setWildcardMatch( aRes, nStartOffset, nEndOffset);
1492
0
                return aRes;
1493
0
            }
1494
1495
0
            nLastAsterisk = nPattern;   // Remember last encountered '*'.
1496
1497
            // cPattern will be the previous non-'*' character, nPattern
1498
            // decremented.
1499
0
            cPattern = rReversePattern.iterateCodePoints( &nPattern, -1);
1500
0
            if (cPattern == mcWildcardEscapeChar && mcWildcardEscapeChar && nPattern > 0)
1501
0
            {
1502
0
                bEscaped = true;
1503
0
                cPattern = rReversePattern.iterateCodePoints( &nPattern, -1);
1504
0
            }
1505
1506
0
            cPatternAfterAsterisk = cPattern;
1507
0
            bEscapedAfterAsterisk = bEscaped;
1508
0
            nPat = nPattern;    // Remember position of pattern before '*', already decremented.
1509
0
            nStr = nString;     // Remember the current string to be matched.
1510
0
        }
1511
1512
0
        if (nString <= nEndPos)
1513
            // Whatever leads in pattern, string will not match.
1514
0
            return aRes;
1515
1516
        // nString will be decremented by iterateCodePoints().
1517
0
        sal_uInt32 cString = searchStr.iterateCodePoints( &nString, -1);
1518
1519
0
        if ((cPattern != '?' || bEscaped) && cPattern != cString)
1520
0
        {
1521
0
            if (nPat == -1)
1522
                // Non-match already without any '*' pattern.
1523
0
                return aRes;
1524
1525
0
            bRewind = true;
1526
0
            nPattern = nPat;                    // Rewind pattern to character before '*', already decremented.
1527
0
            cPattern = cPatternAfterAsterisk;
1528
0
            bEscaped = bEscapedAfterAsterisk;
1529
0
            searchStr.iterateCodePoints( &nStr, -1);
1530
0
            nString = nStr;                     // Restore decremented remembered string position.
1531
0
            if (nPat == nAfterFakePattern)
1532
0
            {
1533
                // Next start offset will be this character (exclusive).
1534
0
                nStartOffset = nString;
1535
0
            }
1536
0
        }
1537
0
        else
1538
0
        {
1539
            // An unescaped '?' pattern matched any character, or characters
1540
            // matched. Reset only escaped state.
1541
0
            bEscaped = false;
1542
0
        }
1543
0
    }
1544
0
    while (nString > nEndPos);
1545
1546
0
    if (bRewind)
1547
0
        return aRes;
1548
1549
    // Eat leading '*' pattern that matches anything, including nothing.
1550
    // '*' is one code unit, so not using iterateCodePoints() is ok.
1551
0
    while (nPattern > 0 && rReversePattern[nPattern-1] == '*')
1552
0
        --nPattern;
1553
1554
0
    if (nPattern == 0)
1555
0
        setWildcardMatch( aRes, nStartOffset, nEndOffset);
1556
0
    return aRes;
1557
0
}
1558
1559
1560
OUString SAL_CALL
1561
TextSearch::getImplementationName()
1562
0
{
1563
0
    return u"com.sun.star.util.TextSearch_i18n"_ustr;
1564
0
}
1565
1566
sal_Bool SAL_CALL TextSearch::supportsService(const OUString& rServiceName)
1567
0
{
1568
0
    return cppu::supportsService(this, rServiceName);
1569
0
}
1570
1571
Sequence< OUString > SAL_CALL
1572
TextSearch::getSupportedServiceNames()
1573
0
{
1574
0
    return { u"com.sun.star.util.TextSearch"_ustr, u"com.sun.star.util.TextSearch2"_ustr };
1575
0
}
1576
1577
extern "C" SAL_DLLPUBLIC_EXPORT css::uno::XInterface*
1578
i18npool_TextSearch_get_implementation(
1579
    css::uno::XComponentContext* context , css::uno::Sequence<css::uno::Any> const&)
1580
3
{
1581
3
    return cppu::acquire(new TextSearch(context));
1582
3
}
1583
1584
/* vim:set shiftwidth=4 softtabstop=4 expandtab: */