Coverage Report

Created: 2026-09-03 06:22

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/espeak-ng/src/libespeak-ng/dictionary.c
Line
Count
Source
1
/*
2
 * Copyright (C) 2005 to 2014 by Jonathan Duddington
3
 * email: jonsd@users.sourceforge.net
4
 * Copyright (C) 2013-2017 Reece H. Dunn
5
 *
6
 * This program is free software; you can redistribute it and/or modify
7
 * it under the terms of the GNU General Public License as published by
8
 * the Free Software Foundation; either version 3 of the License, or
9
 * (at your option) any later version.
10
 *
11
 * This program is distributed in the hope that it will be useful,
12
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
14
 * GNU General Public License for more details.
15
 *
16
 * You should have received a copy of the GNU General Public License
17
 * along with this program; if not, see: <http://www.gnu.org/licenses/>.
18
 */
19
20
#include "config.h"
21
22
#include <ctype.h>
23
#include <stdint.h>
24
#include <stdio.h>
25
#include <stdlib.h>
26
#include <string.h>
27
#include <wctype.h>
28
#include <wchar.h>
29
#include <assert.h>
30
31
#include <espeak-ng/espeak_ng.h>
32
#include <espeak-ng/speak_lib.h>
33
#include <espeak-ng/encoding.h>
34
35
#include "common.h"                // for GetFileLength, strncpy0
36
#include "dictionary.h"
37
#include "numbers.h"                       // for LookupAccentedLetter, Look...
38
#include "phoneme.h"                       // for PHONEME_TAB, phVOWEL, phon...
39
#include "readclause.h"                    // for WordToString2
40
#include "speech.h"                        // for path_home
41
#include "compiledict.h"                   // for DecodeRule
42
#include "synthdata.h"                     // for PhonemeCode, InterpretPhoneme
43
#include "synthesize.h"                    // for STRESS_IS_PRIMARY, phoneme...
44
#include "translate.h"                     // for Translator, utf8_in, LANGU...
45
46
static int LookupFlags(Translator *tr, const char *word, unsigned int flags_out[2]);
47
static void DollarRule(char *word[], char *word_start, int consumed, int group_length, char word_buf[N_WORD_BYTES], Translator *tr, int command, int *failed, int *add_points);
48
49
typedef struct {
50
  int points;
51
  const char *phonemes;
52
  int end_type;
53
  char *del_fwd;
54
} MatchRecord;
55
56
57
int dictionary_skipwords;
58
char dictionary_name[40];
59
60
// accented characters which indicate (in some languages) the start of a separate syllable
61
static const unsigned short diereses_list[7] = { 0xe4, 0xeb, 0xef, 0xf6, 0xfc, 0xff, 0 };
62
63
// convert characters to an approximate 7 bit ascii equivalent
64
// used for checking for vowels (up to 0x259=schwa)
65
460k
#define N_REMOVE_ACCENT  0x25e
66
static const unsigned char remove_accent[N_REMOVE_ACCENT] = {
67
  'a', 'a', 'a', 'a', 'a', 'a', 'a', 'c', 'e', 'e', 'e', 'e', 'i', 'i', 'i', 'i',  // 0c0
68
  'd', 'n', 'o', 'o', 'o', 'o', 'o',   0, 'o', 'u', 'u', 'u', 'u', 'y', 't', 's',  // 0d0
69
  'a', 'a', 'a', 'a', 'a', 'a', 'a', 'c', 'e', 'e', 'e', 'e', 'i', 'i', 'i', 'i',  // 0e0
70
  'd', 'n', 'o', 'o', 'o', 'o', 'o',   0, 'o', 'u', 'u', 'u', 'u', 'y', 't', 'y',  // 0f0
71
72
  'a', 'a', 'a', 'a', 'a', 'a', 'c', 'c', 'c', 'c', 'c', 'c', 'c', 'c', 'd', 'd',  // 100
73
  'd', 'd', 'e', 'e', 'e', 'e', 'e', 'e', 'e', 'e', 'e', 'e', 'g', 'g', 'g', 'g',  // 110
74
  'g', 'g', 'g', 'g', 'h', 'h', 'h', 'h', 'i', 'i', 'i', 'i', 'i', 'i', 'i', 'i',  // 120
75
  'i', 'i', 'i', 'i', 'j', 'j', 'k', 'k', 'k', 'l', 'l', 'l', 'l', 'l', 'l', 'l',  // 130
76
  'l', 'l', 'l', 'n', 'n', 'n', 'n', 'n', 'n', 'n', 'n', 'n', 'o', 'o', 'o', 'o',  // 140
77
  'o', 'o', 'o', 'o', 'r', 'r', 'r', 'r', 'r', 'r', 's', 's', 's', 's', 's', 's',  // 150
78
  's', 's', 't', 't', 't', 't', 't', 't', 'u', 'u', 'u', 'u', 'u', 'u', 'u', 'u',  // 160
79
  'u', 'u', 'u', 'u', 'w', 'w', 'y', 'y', 'y', 'z', 'z', 'z', 'z', 'z', 'z', 's',  // 170
80
  'b', 'b', 'b', 'b',   0,   0, 'o', 'c', 'c', 'd', 'd', 'd', 'd', 'd', 'e', 'e',  // 180
81
  'e', 'f', 'f', 'g', 'g', 'h', 'i', 'i', 'k', 'k', 'l', 'l', 'm', 'n', 'n', 'o',  // 190
82
  'o', 'o', 'o', 'o', 'p', 'p', 'y',   0,   0, 's', 's', 't', 't', 't', 't', 'u',  // 1a0
83
  'u', 'u', 'v', 'y', 'y', 'z', 'z', 'z', 'z', 'z', 'z', 'z',   0,   0,   0, 'w',  // 1b0
84
  't', 't', 't', 'k', 'd', 'd', 'd', 'l', 'l', 'l', 'n', 'n', 'n', 'a', 'a', 'i',  // 1c0
85
  'i', 'o', 'o', 'u', 'u', 'u', 'u', 'u', 'u', 'u', 'u', 'u', 'u', 'e', 'a', 'a',  // 1d0
86
  'a', 'a', 'a', 'a', 'g', 'g', 'g', 'g', 'k', 'k', 'o', 'o', 'o', 'o', 'z', 'z',  // 1e0
87
  'j', 'd', 'd', 'd', 'g', 'g', 'w', 'w', 'n', 'n', 'a', 'a', 'a', 'a', 'o', 'o',  // 1f0
88
89
  'a', 'a', 'a', 'a', 'e', 'e', 'e', 'e', 'i', 'i', 'i', 'i', 'o', 'o', 'o', 'o',  // 200
90
  'r', 'r', 'r', 'r', 'u', 'u', 'u', 'u', 's', 's', 't', 't', 'y', 'y', 'h', 'h',  // 210
91
  'n', 'd', 'o', 'o', 'z', 'z', 'a', 'a', 'e', 'e', 'o', 'o', 'o', 'o', 'o', 'o',  // 220
92
  'o', 'o', 'y', 'y', 'l', 'n', 't', 'j', 'd', 'q', 'a', 'c', 'c', 'l', 't', 's',  // 230
93
  'z',   0,   0, 'b', 'u', 'v', 'e', 'e', 'j', 'j', 'q', 'q', 'r', 'r', 'y', 'y',  // 240
94
  'a', 'a', 'a', 'b', 'o', 'c', 'd', 'd', 'e', 'e', 'e', 'e', 'e', 'e'
95
};
96
97
static int Reverse4Bytes(int word)
98
69.3k
{
99
  // reverse the order of bytes from little-endian to big-endian
100
#ifdef ARCH_BIG
101
  int ix;
102
  int word2 = 0;
103
104
  for (ix = 0; ix <= 24; ix += 8) {
105
    word2 = word2 << 8;
106
    word2 |= (word >> ix) & 0xff;
107
  }
108
  return word2;
109
#else
110
69.3k
  return word;
111
69.3k
#endif
112
69.3k
}
113
114
static void InitGroups(Translator *tr)
115
34.6k
{
116
  // Called after dictionary 1 is loaded, to set up table of entry points for translation rule chains
117
  // for single-letters and two-letter combinations
118
119
34.6k
  int ix;
120
34.6k
  char *p;
121
34.6k
  char *p_name;
122
34.6k
  unsigned char c, c2;
123
34.6k
  int len;
124
125
34.6k
  tr->n_groups2 = 0;
126
8.91M
  for (ix = 0; ix < 256; ix++) {
127
8.87M
    tr->groups1[ix] = NULL;
128
8.87M
    tr->groups2_count[ix] = 0;
129
8.87M
    tr->groups2_start[ix] = 255; // indicates "not set"
130
8.87M
  }
131
34.6k
  memset(tr->letterGroups, 0, sizeof(tr->letterGroups));
132
34.6k
  memset(tr->groups3, 0, sizeof(tr->groups3));
133
134
34.6k
  p = tr->data_dictrules;
135
  // If there are no rules in the dictionary, compile_dictrules will not
136
  // write a RULE_GROUP_START (written in the for loop), but will write
137
  // a RULE_GROUP_END.
138
2.64M
  if (*p != RULE_GROUP_END) while (*p != 0) {
139
2.61M
    if (*p != RULE_GROUP_START) {
140
0
      fprintf(stderr, "Bad rules data in '%s_dict' at 0x%x (%c)\n", dictionary_name, (unsigned int)(p - tr->data_dictrules), *p);
141
0
      break;
142
0
    }
143
2.61M
    p++;
144
145
2.61M
    if (p[0] == RULE_REPLACEMENTS) {
146
30.0k
      p = (char *)(((intptr_t)p+4) & ~3); // advance to next word boundary
147
30.0k
      tr->langopts.replace_chars = (unsigned char *)p;
148
149
3.58M
      while ( !is_str_totally_null(p, 4) ) {
150
3.55M
        p++;
151
3.55M
      }
152
153
176k
      while (*p != RULE_GROUP_END) p++;
154
30.0k
      p++;
155
30.0k
      continue;
156
30.0k
    }
157
158
2.58M
    if (p[0] == RULE_LETTERGP2) {
159
270k
      ix = p[1] - 'A';
160
270k
      if (ix < 0)
161
35.7k
        ix += 256;
162
270k
      p += 2;
163
270k
      if ((ix >= 0) && (ix < N_LETTER_GROUPS))
164
270k
        tr->letterGroups[ix] = p;
165
2.31M
    } else {
166
2.31M
      len = strlen(p);
167
2.31M
      p_name = p;
168
2.31M
      c = p_name[0];
169
2.31M
      c2 = p_name[1];
170
171
2.31M
      p += (len+1);
172
2.31M
      if (len == 1)
173
534k
        tr->groups1[c] = p;
174
1.77M
      else if (len == 0)
175
31.1k
        tr->groups1[0] = p;
176
1.74M
      else if (c == 1) {
177
        // index by offset from letter base
178
743k
        tr->groups3[c2 - 1] = p;
179
1.00M
      } else {
180
1.00M
        if (tr->groups2_start[c] == 255)
181
218k
          tr->groups2_start[c] = tr->n_groups2;
182
183
1.00M
        tr->groups2_count[c]++;
184
1.00M
        tr->groups2[tr->n_groups2] = p;
185
1.00M
        tr->groups2_name[tr->n_groups2++] = (c + (c2 << 8));
186
1.00M
      }
187
2.31M
    }
188
189
    // skip over all the rules in this group
190
95.7M
    while (*p != RULE_GROUP_END)
191
93.1M
      p += (strlen(p) + 1);
192
2.58M
    p++;
193
2.58M
  }
194
34.6k
}
195
196
int LoadDictionary(Translator *tr, const char *name, int no_error)
197
34.6k
{
198
34.6k
  int hash;
199
34.6k
  char *p;
200
34.6k
  int *pw;
201
34.6k
  int length;
202
34.6k
  FILE *f;
203
34.6k
  int size;
204
34.6k
  char fname[N_PATH_BUF];
205
206
34.6k
  if (dictionary_name != name)
207
34.6k
    snprintf(dictionary_name, sizeof(dictionary_name), "%s", name); // currently loaded dictionary name
208
34.6k
  if (tr->dictionary_name != name)
209
9.15k
    snprintf(tr->dictionary_name, sizeof(tr->dictionary_name), "%s", name);
210
211
  // Load a pronunciation data file into memory
212
  // bytes 0-3:  offset to rules data
213
  // bytes 4-7:  number of hash table entries
214
34.6k
  snprintf(fname, sizeof(fname), "%s%c%s_dict", path_home, PATHSEP, name);
215
34.6k
  size = GetFileLength(fname);
216
217
34.6k
  if (tr->data_dictlist != NULL) {
218
0
    free(tr->data_dictlist);
219
0
    tr->data_dictlist = NULL;
220
0
  }
221
222
34.6k
  f = fopen(fname, "rb");
223
34.6k
  if ((f == NULL) || (size <= 0)) {
224
0
    if (no_error == 0)
225
0
      fprintf(stderr, "Can't read dictionary file: '%s'\n", fname);
226
0
    if (f != NULL)
227
0
      fclose(f);
228
0
    return 1;
229
0
  }
230
231
34.6k
  if ((tr->data_dictlist = malloc(size)) == NULL) {
232
0
    fclose(f);
233
0
    return 3;
234
0
  }
235
34.6k
  size = fread(tr->data_dictlist, 1, size, f);
236
34.6k
  fclose(f);
237
238
34.6k
  pw = (int *)(tr->data_dictlist);
239
34.6k
  length = Reverse4Bytes(pw[1]);
240
241
34.6k
  if (size <= (N_HASH_DICT + sizeof(int)*2)) {
242
0
    fprintf(stderr, "Empty _dict file: '%s\n", fname);
243
0
    return 2;
244
0
  }
245
246
34.6k
  if ((Reverse4Bytes(pw[0]) != N_HASH_DICT) ||
247
34.6k
      (length <= 0) || (length > 0x8000000)) {
248
0
    fprintf(stderr, "Bad data: '%s' (%x length=%x)\n", fname, Reverse4Bytes(pw[0]), length);
249
0
    return 2;
250
0
  }
251
34.6k
  tr->data_dictrules = &(tr->data_dictlist[length]);
252
253
  // set up indices into data_dictrules
254
34.6k
  InitGroups(tr);
255
256
  // set up hash table for data_dictlist
257
34.6k
  p = &(tr->data_dictlist[8]);
258
259
35.5M
  for (hash = 0; hash < N_HASH_DICT; hash++) {
260
35.5M
    tr->dict_hashtab[hash] = p;
261
546M
    while ((length = *(uint8_t *)p) != 0)
262
510M
      p += length;
263
35.5M
    p++; // skip over the zero which terminates the list for this hash value
264
35.5M
  }
265
266
34.6k
  if ((tr->dict_min_size > 0) && (size < (unsigned int)tr->dict_min_size))
267
0
    fprintf(stderr, "Full dictionary is not installed for '%s'\n", name);
268
269
34.6k
  return 0;
270
34.6k
}
271
272
/* Generate a hash code from the specified string
273
    This is used to access the dictionary_2 word-lookup dictionary
274
 */
275
int HashDictionary(const char *string)
276
5.47M
{
277
5.47M
  int c;
278
5.47M
  int chars = 0;
279
5.47M
  int hash = 0;
280
281
18.7M
  while ((c = (*string++ & 0xff)) != 0) {
282
13.2M
    hash = hash * 8 + c;
283
13.2M
    hash = (hash & 0x3ff) ^ (hash >> 8); // exclusive or
284
13.2M
    chars++;
285
13.2M
  }
286
287
5.47M
  return (hash+chars) & 0x3ff; // a 10 bit hash code
288
5.47M
}
289
290
/* Translate a phoneme string from ascii mnemonics to internal phoneme numbers,
291
   from 'p' up to next blank .
292
   Returns advanced 'p'
293
   outptr contains encoded phonemes, unrecognized phoneme stops the encoding
294
   bad_phoneme must point to char array of length 2 of more
295
 */
296
const char *EncodePhonemes(const char *p, char *outptr, int *bad_phoneme)
297
133k
{
298
133k
  int ix;
299
133k
  unsigned char c;
300
133k
  int count;     // num. of matching characters
301
133k
  int max;       // highest num. of matching found so far
302
133k
  int max_ph;    // corresponding phoneme with highest matching
303
133k
  int consumed;
304
133k
  unsigned int mnemonic_word;
305
306
133k
  if (bad_phoneme != NULL)
307
100k
    *bad_phoneme = 0;
308
309
  // skip initial blanks
310
133k
  while ((uint8_t)*p < 0x80 && isspace(*p))
311
6
    p++;
312
313
398k
  while (((c = *p) != 0) && !isspace(c)) {
314
350k
    consumed = 0;
315
316
350k
    switch (c)
317
350k
    {
318
1.31k
    case '|':
319
      // used to separate phoneme mnemonics if needed, to prevent characters being treated
320
      // as a multi-letter mnemonic
321
322
1.31k
      if ((c = p[1]) == '|') {
323
        // treat double || as a word-break symbol, drop through
324
        // to the default case with c = '|'
325
1.16k
      } else {
326
151
        p++;
327
151
        break;
328
151
      }
329
349k
    default:
330
      // lookup the phoneme mnemonic, find the phoneme with the highest number of
331
      // matching characters
332
349k
      max = -1;
333
349k
      max_ph = 0;
334
335
47.7M
      for (ix = 1; ix < n_phoneme_tab; ix++) {
336
47.3M
        if (phoneme_tab[ix] == NULL)
337
46
          continue;
338
47.3M
        if (phoneme_tab[ix]->type == phINVALID)
339
0
          continue; // this phoneme is not defined for this language
340
341
47.3M
        count = 0;
342
47.3M
        mnemonic_word = phoneme_tab[ix]->mnemonic;
343
344
48.3M
        while (((c = p[count]) > ' ') && (count < 4) &&
345
48.1M
               (c == ((mnemonic_word >> (count*8)) & 0xff)))
346
1.02M
          count++;
347
348
47.3M
        if ((count > max) &&
349
21.4M
            ((count == 4) || (((mnemonic_word >> (count*8)) & 0xff) == 0))) {
350
276k
          max = count;
351
276k
          max_ph = phoneme_tab[ix]->code;
352
276k
        }
353
47.3M
      }
354
355
349k
      if (max_ph == 0) {
356
        // not recognised, report and ignore
357
85.0k
        if (bad_phoneme != NULL)
358
85.0k
          utf8_in(bad_phoneme, p);
359
85.0k
        *outptr++ = 0;
360
85.0k
        return p+1;
361
85.0k
      }
362
363
264k
      if (max <= 0)
364
0
        max = 1;
365
264k
      p += (consumed + max);
366
264k
      *outptr++ = (char)(max_ph);
367
368
264k
      if (max_ph == phonSWITCH) {
369
        // Switch Language: this phoneme is followed by a text string
370
388
        char *p_lang = outptr;
371
2.38k
        while (!isspace(c = *p) && (c != 0)) {
372
1.99k
          p++;
373
1.99k
          *outptr++ = tolower(c);
374
1.99k
        }
375
388
        *outptr = 0;
376
388
        if (c == 0) {
377
0
          if (strcmp(p_lang, ESPEAKNG_DEFAULT_VOICE) == 0) {
378
0
            *p_lang = 0; // don't need ESPEAKNG_DEFAULT_VOICE, it's assumed by default
379
0
            return p;
380
0
          }
381
0
        } else
382
388
          *outptr++ = '|'; // more phonemes follow, terminate language string with separator
383
388
      }
384
264k
      break;
385
350k
    }
386
350k
  }
387
  // terminate the encoded string
388
48.6k
  *outptr = 0;
389
48.6k
  return p;
390
133k
}
391
392
void DecodePhonemes(const char *inptr, char *outptr)
393
18.3k
{
394
  // Translate from internal phoneme codes into phoneme mnemonics
395
18.3k
  unsigned char phcode;
396
18.3k
  unsigned char c;
397
18.3k
  unsigned int mnem;
398
18.3k
  PHONEME_TAB *ph;
399
18.3k
  static const char stress_chars[] = "==,,'*  ";
400
401
18.3k
  sprintf(outptr, "* ");
402
187k
  while ((phcode = *inptr++) > 0) {
403
168k
    if (phcode == 255)
404
0
      continue; // indicates unrecognised phoneme
405
168k
    if ((ph = phoneme_tab[phcode]) == NULL)
406
0
      continue;
407
408
168k
    if ((ph->type == phSTRESS) && (ph->std_length <= 4) && (ph->program == 0)) {
409
34.1k
      if (ph->std_length > 1)
410
23.5k
        *outptr++ = stress_chars[ph->std_length];
411
134k
    } else {
412
134k
      mnem = ph->mnemonic;
413
414
286k
      while ((c = (mnem & 0xff)) != 0) {
415
151k
        *outptr++ = c;
416
151k
        mnem = mnem >> 8;
417
151k
      }
418
134k
      if (phcode == phonSWITCH) {
419
178
        while (isalpha(*inptr))
420
178
          *outptr++ = *inptr++;
421
178
      }
422
134k
    }
423
168k
  }
424
18.3k
  *outptr = 0; // string terminator
425
18.3k
}
426
427
// using Kirschenbaum to IPA translation, ascii 0x20 to 0x7f
428
static const unsigned short ipa1[96] = {
429
  0x20,  0x21,  0x22,  0x2b0, 0x24,  0x25,  0x0e6, 0x2c8, 0x28,  0x29,  0x27e, 0x2b,  0x2cc, 0x2d,  0x2e,  0x2f,
430
  0x252, 0x31,  0x32,  0x25c, 0x34,  0x35,  0x36,  0x37,  0x275, 0x39,  0x2d0, 0x2b2, 0x3c,  0x3d,  0x3e,  0x294,
431
  0x259, 0x251, 0x3b2, 0xe7,  0xf0,  0x25b, 0x46,  0x262, 0x127, 0x26a, 0x25f, 0x4b,  0x26b, 0x271, 0x14b, 0x254,
432
  0x3a6, 0x263, 0x280, 0x283, 0x3b8, 0x28a, 0x28c, 0x153, 0x3c7, 0xf8,  0x292, 0x32a, 0x5c,  0x5d,  0x5e,  0x5f,
433
  0x60,  0x61,  0x62,  0x63,  0x64,  0x65,  0x66,  0x261, 0x68,  0x69,  0x6a,  0x6b,  0x6c,  0x6d,  0x6e,  0x6f,
434
  0x70,  0x71,  0x72,  0x73,  0x74,  0x75,  0x76,  0x77,  0x78,  0x79,  0x7a,  0x7b,  0x7c,  0x7d,  0x303, 0x7f
435
};
436
437
0
#define N_PHON_OUT  500  // realloc increment
438
static char *phon_out_buf = NULL;   // passes the result of GetTranslatedPhonemeString()
439
static unsigned int phon_out_size = 0;
440
441
char *WritePhMnemonic(char *phon_out, PHONEME_TAB *ph, PHONEME_LIST *plist, int use_ipa, int *flags)
442
0
{
443
0
  int c;
444
0
  int mnem;
445
0
  int len;
446
0
  bool first;
447
0
  int ix = 0;
448
0
  char *p;
449
0
  PHONEME_DATA phdata;
450
451
0
  if (ph->code == phonEND_WORD) {
452
    // ignore
453
0
    phon_out[0] = 0;
454
0
    return phon_out;
455
0
  }
456
457
0
  if (ph->code == phonSWITCH) {
458
    // the tone_ph field contains a phoneme table number
459
0
    p = phoneme_tab_list[plist->tone_ph].name;
460
0
    sprintf(phon_out, "(%s)", p);
461
0
    return phon_out + strlen(phon_out);
462
0
  }
463
464
0
  if (use_ipa) {
465
    // has an ipa name been defined for this phoneme ?
466
0
    phdata.ipa_string[0] = 0;
467
468
0
    if (plist == NULL)
469
0
      InterpretPhoneme2(ph->code, &phdata);
470
0
    else
471
0
      InterpretPhoneme(NULL, 0, plist, phoneme_list, &phdata, NULL);
472
473
0
    p = phdata.ipa_string;
474
0
    if (*p == 0x20) {
475
      // indicates no name for this phoneme
476
0
      *phon_out = 0;
477
0
      return phon_out;
478
0
    }
479
0
    if ((*p != 0) && ((*p & 0xff) < 0x20)) {
480
      // name starts with a flags byte
481
0
      if (flags != NULL)
482
0
        *flags = *p;
483
0
      p++;
484
0
    }
485
486
0
    len = strlen(p);
487
0
    if (len > 0) {
488
0
      strcpy(phon_out, p);
489
0
      phon_out += len;
490
0
      *phon_out = 0;
491
0
      return phon_out;
492
0
    }
493
0
  }
494
495
0
  first = true;
496
0
  for (mnem = ph->mnemonic; (c = mnem & 0xff) != 0; mnem = mnem >> 8) {
497
0
    if (c == '/')
498
0
      break; // discard phoneme variant indicator
499
500
0
    if (use_ipa) {
501
      // convert from ascii to ipa
502
0
      if (first && (c == '_'))
503
0
        break; // don't show pause phonemes
504
505
0
      if ((c == '#') && (ph->type == phVOWEL))
506
0
        break; // # is subscript-h, but only for consonants
507
508
      // ignore digits after the first character
509
0
      if (!first && IsDigit09(c))
510
0
        continue;
511
512
0
      if ((c >= 0x20) && (c < 128))
513
0
        c = ipa1[c-0x20];
514
515
0
      ix += utf8_out(c, &phon_out[ix]);
516
0
    } else
517
0
      phon_out[ix++] = c;
518
0
    first = false;
519
0
  }
520
521
0
  phon_out = &phon_out[ix];
522
0
  *phon_out = 0;
523
0
  return phon_out;
524
0
}
525
526
//// Extension: write phone mnemonic with stress
527
0
char *WritePhMnemonicWithStress(char *phon_out, PHONEME_TAB *ph, PHONEME_LIST *plist, int use_ipa, int *flags) {
528
0
  if (plist->synthflags & SFLAG_SYLLABLE) {
529
0
    unsigned char stress = plist->stresslevel;
530
531
0
    if (stress > 1) {
532
0
      int c = 0;
533
534
0
      if (stress > STRESS_IS_PRIORITY) {
535
0
        stress = STRESS_IS_PRIORITY;
536
0
      }
537
538
0
      if (use_ipa) {
539
0
        c = 0x2cc; // ipa, secondary stress
540
541
0
        if (stress > STRESS_IS_SECONDARY) {
542
0
          c = 0x02c8; // ipa, primary stress
543
0
        }
544
0
      } else {
545
0
        const char stress_chars[] = "==,,''";
546
547
0
        c = stress_chars[stress];
548
0
      }
549
550
0
      if (c != 0) {
551
0
        phon_out += utf8_out(c, phon_out);
552
0
      }
553
0
    }
554
0
  }
555
556
0
  return WritePhMnemonic(phon_out, ph, plist, use_ipa, flags);
557
0
}
558
////
559
560
const char *GetTranslatedPhonemeString(int phoneme_mode)
561
0
{
562
  /* Called after a clause has been translated into phonemes, in order
563
     to display the clause in phoneme mnemonic form.
564
565
     phoneme_mode
566
                   bit  1:   use IPA phoneme names
567
                   bit  7:   use tie between letters in multi-character phoneme names
568
                   bits 8-23 tie or separator character
569
570
   */
571
572
0
  int ix;
573
0
  unsigned int len;
574
0
  int phon_out_ix = 0;
575
0
  int stress;
576
0
  int c;
577
0
  char *p;
578
0
  char *buf;
579
0
  int count;
580
0
  int flags;
581
0
  int use_ipa;
582
0
  int use_tie;
583
0
  int separate_phonemes;
584
0
  char phon_buf[30];
585
0
  char phon_buf2[30];
586
0
  PHONEME_LIST *plist;
587
588
0
  static const char stress_chars[] = "==,,''";
589
590
0
  if (phon_out_buf == NULL) {
591
0
    phon_out_size = N_PHON_OUT;
592
0
    if ((phon_out_buf = (char *)malloc(phon_out_size)) == NULL) {
593
0
      phon_out_size = 0;
594
0
      return "";
595
0
    }
596
0
  }
597
598
0
  use_ipa = phoneme_mode & espeakPHONEMES_IPA;
599
0
  if (phoneme_mode & espeakPHONEMES_TIE) {
600
0
    use_tie = phoneme_mode >> 8;
601
0
    separate_phonemes = 0;
602
0
  } else {
603
0
    separate_phonemes = phoneme_mode >> 8;
604
0
    use_tie = 0;
605
0
  }
606
607
0
  for (ix = 1; ix < (n_phoneme_list-2); ix++) {
608
0
    buf = phon_buf;
609
610
0
    plist = &phoneme_list[ix];
611
612
0
    WritePhMnemonic(phon_buf2, plist->ph, plist, use_ipa, &flags);
613
0
    if (plist->newword & PHLIST_START_OF_WORD && !(plist->newword & (PHLIST_START_OF_SENTENCE | PHLIST_START_OF_CLAUSE)))
614
0
      *buf++ = ' ';
615
616
0
    if ((!plist->newword) || (separate_phonemes == ' ')) {
617
0
      if ((separate_phonemes != 0) && (ix > 1)) {
618
0
        utf8_in(&c, phon_buf2);
619
0
        if ((c < 0x2b0) || (c > 0x36f)) // not if the phoneme starts with a superscript letter
620
0
          buf += utf8_out(separate_phonemes, buf);
621
0
      }
622
0
    }
623
624
0
    if (plist->synthflags & SFLAG_SYLLABLE) {
625
0
      if ((stress = plist->stresslevel) > 1) {
626
0
        c = 0;
627
0
        if (stress > STRESS_IS_PRIORITY) stress = STRESS_IS_PRIORITY;
628
629
0
        if (use_ipa) {
630
0
          c = 0x2cc; // ipa, secondary stress
631
0
          if (stress > STRESS_IS_SECONDARY)
632
0
            c = 0x02c8; // ipa, primary stress
633
0
        } else
634
0
          c = stress_chars[stress];
635
636
0
        if (c != 0)
637
0
          buf += utf8_out(c, buf);
638
0
      }
639
0
    }
640
641
0
    flags = 0;
642
0
    count = 0;
643
0
    for (p = phon_buf2; *p != 0;) {
644
0
      p += utf8_in(&c, p);
645
0
      if (use_tie != 0) {
646
        // look for non-initial alphabetic character, but not diacritic, superscript etc.
647
0
        if ((count > 0) && !(flags & (1 << (count-1))) && ((c < 0x2b0) || (c > 0x36f)) && iswalpha(c))
648
0
          buf += utf8_out(use_tie, buf);
649
0
      }
650
0
      buf += utf8_out(c, buf);
651
0
      count++;
652
0
    }
653
654
0
    if (plist->ph->code != phonSWITCH) {
655
0
      if (plist->synthflags & SFLAG_LENGTHEN)
656
0
        buf = WritePhMnemonic(buf, phoneme_tab[phonLENGTHEN], plist, use_ipa, NULL);
657
0
      if ((plist->synthflags & SFLAG_SYLLABLE) && (plist->type != phVOWEL)) {
658
        // syllablic consonant
659
0
        buf = WritePhMnemonic(buf, phoneme_tab[phonSYLLABIC], plist, use_ipa, NULL);
660
0
      }
661
0
      if (plist->tone_ph > 0) {
662
0
        PHONEME_TAB *tone_ph = TonePhoneme(plist);
663
0
        if (tone_ph != NULL)
664
0
          buf = WritePhMnemonic(buf, tone_ph, plist, use_ipa, NULL);
665
0
      }
666
0
    }
667
668
0
    len = buf - phon_buf;
669
0
    if ((phon_out_ix + len) >= phon_out_size) {
670
      // enlarge the phoneme buffer
671
0
      phon_out_size = phon_out_ix + len + N_PHON_OUT;
672
0
      char *new_phon_out_buf = (char *)realloc(phon_out_buf, phon_out_size);
673
0
      if (new_phon_out_buf == NULL) {
674
0
        phon_out_size = 0;
675
0
        return "";
676
0
      } else
677
0
        phon_out_buf = new_phon_out_buf;
678
0
    }
679
680
0
    phon_buf[len] = 0;
681
0
    strcpy(&phon_out_buf[phon_out_ix], phon_buf);
682
0
    phon_out_ix += len;
683
0
  }
684
685
0
  if (!phon_out_buf)
686
0
    return "";
687
688
0
  phon_out_buf[phon_out_ix] = 0;
689
690
0
  return phon_out_buf;
691
0
}
692
693
static int LetterGroupNo(char *rule)
694
14.0M
{
695
  /*
696
   * Returns number of letter group
697
   */
698
14.0M
  int groupNo = *rule;
699
14.0M
  groupNo = groupNo - 'A'; // subtracting 'A' makes letter_group equal to number in .Lxx definition
700
14.0M
  if (groupNo < 0)         // fix sign if necessary
701
1.53k
    groupNo += 256;
702
14.0M
  return groupNo;
703
14.0M
}
704
705
static int IsLetterGroup(Translator *tr, char *word, int group, int pre)
706
8.06M
{
707
  /* Match the word against a list of utf-8 strings.
708
   * returns length of matching letter group or -1
709
   *
710
   * How this works:
711
   *
712
   *       +-+
713
   *       |c|<-(tr->letterGroups[group])
714
   *       |0|
715
   *   *p->|c|<-len+              +-+
716
   *       |s|<----+              |a|<-(Actual word to be tested)
717
   *       |0|            *word-> |t|<-*w=word-len+1 (for pre-rule)
718
   *       |~|                    |a|<-*w=word       (for post-rule)
719
   *       |7|                    |s|
720
   *       +-+                    +-+
721
   *
722
   *     7=RULE_GROUP_END
723
   *     0=null terminator
724
   *     pre==1 — pre-rule
725
   *     pre==0 — post-rule
726
   */
727
8.06M
  char *p; // group counter
728
8.06M
  char *w; // word counter
729
8.06M
  int len = 0, i;
730
731
8.06M
  p = tr->letterGroups[group];
732
8.06M
  if (p == NULL)
733
0
    return -1;
734
735
84.6M
  while (*p != RULE_GROUP_END) {
736
    // If '~' (no character) is allowed in group, return 0.
737
76.8M
    if (*p == '~')
738
53.6k
      return 0;
739
740
76.7M
    if (pre) {
741
28.6M
      len = strlen(p);
742
28.6M
      w = word;
743
28.6M
      if (*w == 0)
744
703
        goto skip;
745
46.9M
      for (i = 0; i < len-1; i++)
746
18.3M
      {
747
18.3M
        w--;
748
18.3M
        if (*w == 0)
749
          // Not found, skip the rest of this group.
750
21.9k
          goto skip;
751
18.3M
      }
752
28.6M
    } else
753
48.0M
      w = word;
754
755
    //  Check current group
756
78.9M
    while ((*p == *w) && (*w != 0)) {
757
2.16M
      w++;
758
2.16M
      p++;
759
2.16M
    }
760
76.7M
    if (*p == 0) { // Matched the current group.
761
175k
      if (pre)
762
26.7k
        return len;
763
148k
      return w - word;
764
175k
    }
765
766
    // No match, so skip the rest of this group.
767
76.5M
skip:
768
206M
    while (*p++ != 0)
769
129M
      ;
770
76.5M
  }
771
  // Not found
772
7.84M
  return -1;
773
8.06M
}
774
775
static int IsLetter(Translator *tr, int letter, int group)
776
10.9M
{
777
10.9M
  int letter2;
778
779
10.9M
  if (tr->letter_groups[group] != NULL) {
780
103k
    if (wcschr(tr->letter_groups[group], letter))
781
14.6k
      return 1;
782
89.3k
    return 0;
783
103k
  }
784
785
10.8M
  if (group > 7)
786
0
    return 0;
787
788
10.8M
  if (tr->letter_bits_offset > 0) {
789
54.6k
    if (((letter2 = (letter - tr->letter_bits_offset)) > 0) && (letter2 < 0x100))
790
40.6k
      letter = letter2;
791
14.0k
    else
792
14.0k
      return 0;
793
10.8M
  } else if ((letter >= 0xc0) && (letter < N_REMOVE_ACCENT))
794
30.4k
    return tr->letter_bits[remove_accent[letter-0xc0]] & (1L << group);
795
796
10.8M
  if ((letter >= 0) && (letter < 0x100))
797
10.7M
    return tr->letter_bits[letter] & (1L << group);
798
799
70.1k
  return 0;
800
10.8M
}
801
802
int IsVowel(Translator *tr, int letter)
803
3.86M
{
804
3.86M
  return IsLetter(tr, letter, LETTERGP_VOWEL2);
805
3.86M
}
806
807
int GetVowelStress(Translator *tr, unsigned char *phonemes, signed char *vowel_stress, int *vowel_count, int *stressed_syllable, int control)
808
901k
{
809
  // control = 1, set stress to 1 for forced unstressed vowels
810
901k
  unsigned char phcode;
811
901k
  PHONEME_TAB *ph;
812
901k
  unsigned char *ph_out = phonemes;
813
901k
  int count = 1;
814
901k
  int max_stress = -1;
815
901k
  int ix;
816
901k
  int j;
817
901k
  int stress = -1;
818
901k
  int primary_posn = 0;
819
820
901k
  vowel_stress[0] = STRESS_IS_UNSTRESSED;
821
6.12M
  while (((phcode = *phonemes++) != 0) && (count < (N_WORD_PHONEMES/2)-1)) {
822
5.22M
    if ((ph = phoneme_tab[phcode]) == NULL)
823
22
      continue;
824
825
5.22M
    if ((ph->type == phSTRESS) && (ph->program == 0)) {
826
      // stress marker, use this for the following vowel
827
828
488k
      if (phcode == phonSTRESS_PREV) {
829
        // primary stress on preceding vowel
830
6.19k
        j = count - 1;
831
6.19k
        while ((j > 0) && (*stressed_syllable == 0) && (vowel_stress[j] < STRESS_IS_PRIMARY)) {
832
2.46k
          if ((vowel_stress[j] != STRESS_IS_DIMINISHED) && (vowel_stress[j] != STRESS_IS_UNSTRESSED)) {
833
            // don't promote a phoneme which must be unstressed
834
2.45k
            vowel_stress[j] = STRESS_IS_PRIMARY;
835
836
2.45k
            if (max_stress < STRESS_IS_PRIMARY) {
837
2.17k
              max_stress = STRESS_IS_PRIMARY;
838
2.17k
              primary_posn = j;
839
2.17k
            }
840
841
            /* reduce any preceding primary stress markers */
842
3.95k
            for (ix = 1; ix < j; ix++) {
843
1.49k
              if (vowel_stress[ix] == STRESS_IS_PRIMARY)
844
274
                vowel_stress[ix] = STRESS_IS_SECONDARY;
845
1.49k
            }
846
2.45k
            break;
847
2.45k
          }
848
7
          j--;
849
7
        }
850
482k
      } else {
851
482k
        if ((ph->std_length < 4) || (*stressed_syllable == 0)) {
852
482k
          stress = ph->std_length;
853
854
482k
          if (stress > max_stress)
855
317k
            max_stress = stress;
856
482k
        }
857
482k
      }
858
488k
      continue;
859
488k
    }
860
861
4.73M
    if ((ph->type == phVOWEL) && !(ph->phflags & phNONSYLLABIC)) {
862
1.83M
      vowel_stress[count] = (char)stress;
863
1.83M
      if ((stress >= STRESS_IS_PRIMARY) && (stress >= max_stress)) {
864
419k
        primary_posn = count;
865
419k
        max_stress = stress;
866
419k
      }
867
868
1.83M
      if ((stress < 0) && (control & 1) && (ph->phflags & phUNSTRESSED))
869
130k
        vowel_stress[count] = STRESS_IS_UNSTRESSED; // weak vowel, must be unstressed
870
871
1.83M
      count++;
872
1.83M
      stress = -1;
873
2.89M
    } else if (phcode == phonSYLLABIC) {
874
      // previous consonant phoneme is syllablic
875
1.00k
      vowel_stress[count] = (char)stress;
876
1.00k
      if ((stress < 0) && (control & 1))
877
793
        vowel_stress[count] = STRESS_IS_UNSTRESSED; // syllabic consonant, usually unstressed
878
1.00k
      count++;
879
1.00k
    }
880
881
4.73M
    *ph_out++ = phcode;
882
4.73M
  }
883
901k
  vowel_stress[count] = STRESS_IS_UNSTRESSED;
884
901k
  *ph_out = 0;
885
886
  // has the position of the primary stress been specified by $1, $2, etc?
887
901k
  if (*stressed_syllable > 0) {
888
1.14k
    if (*stressed_syllable >= count)
889
0
      *stressed_syllable = count-1; // the final syllable
890
891
1.14k
    vowel_stress[*stressed_syllable] = STRESS_IS_PRIMARY;
892
1.14k
    max_stress = STRESS_IS_PRIMARY;
893
1.14k
    primary_posn = *stressed_syllable;
894
1.14k
  }
895
896
901k
  if (max_stress == STRESS_IS_PRIORITY) {
897
    // priority stress, replaces any other primary stress marker
898
1.10k
    for (ix = 1; ix < count; ix++) {
899
841
      if (vowel_stress[ix] == STRESS_IS_PRIMARY) {
900
49
        if (tr->langopts.stress_flags & S_PRIORITY_STRESS)
901
8
          vowel_stress[ix] = STRESS_IS_UNSTRESSED;
902
41
        else
903
41
          vowel_stress[ix] = STRESS_IS_SECONDARY;
904
49
      }
905
906
841
      if (vowel_stress[ix] == STRESS_IS_PRIORITY) {
907
266
        vowel_stress[ix] = STRESS_IS_PRIMARY;
908
266
        primary_posn = ix;
909
266
      }
910
841
    }
911
266
    max_stress = STRESS_IS_PRIMARY;
912
266
  }
913
914
901k
  *stressed_syllable = primary_posn;
915
901k
  *vowel_count = count;
916
901k
  return max_stress;
917
901k
}
918
919
const char stress_phonemes[] = {
920
  phonSTRESS_D, phonSTRESS_U, phonSTRESS_2, phonSTRESS_3,
921
  phonSTRESS_P, phonSTRESS_P2, phonSTRESS_TONIC
922
};
923
924
void SetWordStress(Translator *tr, char *output, unsigned int *dictionary_flags, int tonic, int control)
925
1.18M
{
926
  /* Guess stress pattern of word.  This is language specific
927
928
     'output' is used for input and output
929
930
     'dictionary_flags' has bits 0-3   position of stressed vowel (if > 0)
931
                                       or unstressed (if == 7) or syllables 1 and 2 (if == 6)
932
                            bits 8...  dictionary flags
933
934
     If 'tonic' is set (>= 0), replace highest stress by this value.
935
936
     control:  bit 0   This is an individual symbol, not a word
937
              bit 1   Suffix phonemes are still to be added
938
   */
939
940
1.18M
  unsigned char phcode;
941
1.18M
  unsigned char *p;
942
1.18M
  PHONEME_TAB *ph;
943
1.18M
  int stress;
944
1.18M
  int max_stress;
945
1.18M
  int max_stress_input; // any stress specified in the input?
946
1.18M
  int vowel_count; // num of vowels + 1
947
1.18M
  int ix;
948
1.18M
  int v;
949
1.18M
  int v_stress;
950
1.18M
  int stressed_syllable; // position of stressed syllable
951
1.18M
  int max_stress_posn;
952
1.18M
  char *max_output;
953
1.18M
  int final_ph;
954
1.18M
  int final_ph2;
955
1.18M
  int mnem;
956
1.18M
  int opt_length;
957
1.18M
  int stressflags;
958
1.18M
  int dflags = 0;
959
1.18M
  int first_primary;
960
1.18M
  int long_vowel;
961
962
1.18M
  signed char vowel_stress[N_WORD_PHONEMES/2];
963
1.18M
  char syllable_weight[N_WORD_PHONEMES/2];
964
1.18M
  char vowel_length[N_WORD_PHONEMES/2];
965
1.18M
  unsigned char phonetic[N_WORD_PHONEMES];
966
967
1.18M
  static const char consonant_types[16] = { 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0 };
968
969
1.18M
  memset(syllable_weight, 0, sizeof(syllable_weight));
970
1.18M
  memset(vowel_length, 0, sizeof(vowel_length));
971
972
1.18M
  stressflags = tr->langopts.stress_flags;
973
974
1.18M
  if (dictionary_flags != NULL)
975
1.18M
    dflags = dictionary_flags[0];
976
977
  // copy input string into internal buffer
978
6.17M
  for (ix = 0; ix < N_WORD_PHONEMES; ix++) {
979
6.17M
    phonetic[ix] = output[ix];
980
6.17M
    if (phonetic[ix] == 0)
981
1.18M
      break;
982
    // check for unknown phoneme codes. A code may be out of range, or it may
983
    // fall in a gap of the current table, in which case phoneme_tab holds NULL
984
    // for it (see SelectPhonemeTable). The code below dereferences these
985
    // entries unconditionally, so both cases must be substituted here.
986
4.99M
    if ((phonetic[ix] >= n_phoneme_tab) || (phoneme_tab[phonetic[ix]] == NULL))
987
15.1k
      phonetic[ix] = phonSCHWA;
988
4.99M
  }
989
1.18M
  if (ix == 0) return;
990
858k
  final_ph = phonetic[ix-1];
991
858k
  final_ph2 = phonetic[(ix > 1) ? ix-2 : ix-1];
992
993
858k
  max_output = output + (N_WORD_PHONEMES-3); // check for overrun
994
995
996
  // any stress position marked in the xx_list dictionary ?
997
858k
  bool unstressed_word = false;
998
858k
  stressed_syllable = dflags & 0x7;
999
858k
  if (dflags & 0x8) {
1000
    // this indicates a word without a primary stress
1001
15.5k
    stressed_syllable = dflags & 0x3;
1002
15.5k
    unstressed_word = true;
1003
15.5k
  }
1004
1005
858k
  max_stress = max_stress_input = GetVowelStress(tr, phonetic, vowel_stress, &vowel_count, &stressed_syllable, 1);
1006
858k
  if ((max_stress < 0) && dictionary_flags)
1007
578k
    max_stress = STRESS_IS_DIMINISHED;
1008
1009
  // heavy or light syllables
1010
858k
  ix = 1;
1011
5.35M
  for (p = phonetic; *p != 0; p++) {
1012
4.49M
    if ((phoneme_tab[p[0]]->type == phVOWEL) && !(phoneme_tab[p[0]]->phflags & phNONSYLLABIC)) {
1013
1.76M
      int weight = 0;
1014
1.76M
      bool lengthened = false;
1015
1016
1.76M
      if (phoneme_tab[p[1]]->code == phonLENGTHEN)
1017
51.1k
        lengthened = true;
1018
1019
1.76M
      if (lengthened || (phoneme_tab[p[0]]->phflags & phLONG)) {
1020
        // long vowel, increase syllable weight
1021
76.2k
        weight++;
1022
76.2k
      }
1023
1.76M
      vowel_length[ix] = weight;
1024
1025
1.76M
      if (lengthened) p++; // advance over phonLENGTHEN
1026
1027
1.76M
      if (consonant_types[phoneme_tab[p[1]]->type] && ((phoneme_tab[p[2]]->type != phVOWEL) || (phoneme_tab[p[1]]->phflags & phLONG))) {
1028
        // followed by two consonants, a long consonant, or consonant and end-of-word
1029
711k
        weight++;
1030
711k
      }
1031
1.76M
      syllable_weight[ix] = weight;
1032
1.76M
      ix++;
1033
1.76M
    }
1034
4.49M
  }
1035
1036
858k
  switch (tr->langopts.stress_rule)
1037
858k
  {
1038
3.58k
  case STRESSPOSN_2LLH:
1039
    // stress on first syllable, unless it is a light syllable followed by a heavy syllable
1040
3.58k
    if ((syllable_weight[1] > 0) || (syllable_weight[2] == 0))
1041
3.01k
      break;
1042
    // fallthrough:
1043
565
  case STRESSPOSN_2L:
1044
    // stress on second syllable
1045
565
    if ((stressed_syllable == 0) && (vowel_count > 2)) {
1046
17
      stressed_syllable = 2;
1047
17
      if (max_stress == STRESS_IS_DIMINISHED)
1048
17
        vowel_stress[stressed_syllable] = STRESS_IS_PRIMARY;
1049
17
      max_stress = STRESS_IS_PRIMARY;
1050
17
    }
1051
565
    break;
1052
1053
93.9k
  case STRESSPOSN_2R:
1054
    // a language with stress on penultimate vowel
1055
1056
93.9k
    if (stressed_syllable == 0) {
1057
      // no explicit stress - stress the penultimate vowel
1058
52.6k
      max_stress = STRESS_IS_PRIMARY;
1059
1060
52.6k
      if (vowel_count > 2) {
1061
17.2k
        stressed_syllable = vowel_count - 2;
1062
1063
17.2k
        if (stressflags & S_FINAL_SPANISH) {
1064
          // LANG=Spanish, stress on last vowel if the word ends in a consonant other than 'n' or 's'
1065
4.63k
          if (phoneme_tab[final_ph]->type != phVOWEL) {
1066
1.50k
            mnem = phoneme_tab[final_ph]->mnemonic;
1067
1068
1.50k
            if ((tr->translator_name == L('a', 'n')) || (tr->translator_name == L('c', 'a'))) {
1069
118
              if (((mnem != 's') && (mnem != 'n')) || phoneme_tab[final_ph2]->type != phVOWEL)
1070
116
                stressed_syllable = vowel_count - 1; // stress on last syllable
1071
1.39k
            } else if (tr->translator_name == L('i', 'a')) {
1072
0
              if ((mnem != 's') || phoneme_tab[final_ph2]->type != phVOWEL)
1073
0
                stressed_syllable = vowel_count - 1; // stress on last syllable
1074
1.39k
            } else {
1075
1.39k
              if ((mnem == 's') && (phoneme_tab[final_ph2]->type == phNASAL)) {
1076
                // -ns  stress remains on penultimate syllable
1077
1.38k
              } else if (((phoneme_tab[final_ph]->type != phNASAL) && (mnem != 's')) || (phoneme_tab[final_ph2]->type != phVOWEL))
1078
467
                stressed_syllable = vowel_count - 1;
1079
1.39k
            }
1080
1.50k
          }
1081
4.63k
        }
1082
1083
17.2k
        if (stressflags & S_FINAL_LONG) {
1084
          // stress on last syllable if it has a long vowel, but previous syllable has a short vowel
1085
2.17k
          if (vowel_length[vowel_count - 1] > vowel_length[vowel_count - 2])
1086
861
            stressed_syllable = vowel_count - 1;
1087
2.17k
        }
1088
1089
17.2k
        if ((vowel_stress[stressed_syllable] == STRESS_IS_DIMINISHED) || (vowel_stress[stressed_syllable] == STRESS_IS_UNSTRESSED)) {
1090
          // but this vowel is explicitly marked as unstressed
1091
395
          if (stressed_syllable > 1)
1092
284
            stressed_syllable--;
1093
111
          else
1094
111
            stressed_syllable++;
1095
395
        }
1096
17.2k
      } else
1097
35.4k
        stressed_syllable = 1;
1098
1099
      // only set the stress if it's not already marked explicitly
1100
52.6k
      if (vowel_stress[stressed_syllable] < 0) {
1101
        // don't stress if next and prev syllables are stressed
1102
49.0k
        if ((vowel_stress[stressed_syllable-1] < STRESS_IS_PRIMARY) || (vowel_stress[stressed_syllable+1] < STRESS_IS_PRIMARY))
1103
49.0k
          vowel_stress[stressed_syllable] = max_stress;
1104
49.0k
      }
1105
52.6k
    }
1106
93.9k
    break;
1107
77.7k
  case STRESSPOSN_1R:
1108
    // stress on last vowel
1109
77.7k
    if (stressed_syllable == 0) {
1110
      // no explicit stress - stress the final vowel
1111
36.7k
      stressed_syllable = vowel_count - 1;
1112
1113
37.5k
      while (stressed_syllable > 0) {
1114
        // find the last vowel which is not unstressed
1115
36.5k
        if (vowel_stress[stressed_syllable] < STRESS_IS_DIMINISHED) {
1116
35.7k
          vowel_stress[stressed_syllable] = STRESS_IS_PRIMARY;
1117
35.7k
          break;
1118
35.7k
        } else
1119
782
          stressed_syllable--;
1120
36.5k
      }
1121
36.7k
      max_stress = STRESS_IS_PRIMARY;
1122
36.7k
    }
1123
77.7k
    break;
1124
33.5k
  case  STRESSPOSN_3R: // stress on antipenultimate vowel
1125
33.5k
    if (stressed_syllable == 0) {
1126
19.5k
      stressed_syllable = vowel_count - 3;
1127
19.5k
      if (stressed_syllable < 1)
1128
16.9k
        stressed_syllable = 1;
1129
1130
19.5k
      if (max_stress == STRESS_IS_DIMINISHED)
1131
19.5k
        vowel_stress[stressed_syllable] = STRESS_IS_PRIMARY;
1132
19.5k
      max_stress = STRESS_IS_PRIMARY;
1133
19.5k
    }
1134
33.5k
    break;
1135
17.3k
  case STRESSPOSN_SYLCOUNT:
1136
    // LANG=Russian
1137
17.3k
    if (stressed_syllable == 0) {
1138
      // no explicit stress - guess the stress from the number of syllables
1139
880
      static const char guess_ru[16] =   { 0, 0, 1, 1, 2, 3, 3, 4, 5, 6, 7, 7, 8, 9, 10, 11 };
1140
880
      static const char guess_ru_v[16] = { 0, 0, 1, 1, 2, 2, 3, 3, 4, 5, 6, 7, 7, 8, 9, 10 }; // for final phoneme is a vowel
1141
880
      static const char guess_ru_t[16] = { 0, 0, 1, 2, 3, 3, 3, 4, 5, 6, 7, 7, 7, 8, 9, 10 }; // for final phoneme is an unvoiced stop
1142
1143
880
      stressed_syllable = vowel_count - 3;
1144
880
      if (vowel_count < 16) {
1145
880
        if (phoneme_tab[final_ph]->type == phVOWEL)
1146
213
          stressed_syllable = guess_ru_v[vowel_count];
1147
667
        else if (phoneme_tab[final_ph]->type == phSTOP)
1148
304
          stressed_syllable = guess_ru_t[vowel_count];
1149
363
        else
1150
363
          stressed_syllable = guess_ru[vowel_count];
1151
880
      }
1152
880
      vowel_stress[stressed_syllable] = STRESS_IS_PRIMARY;
1153
880
      max_stress = STRESS_IS_PRIMARY;
1154
880
    }
1155
17.3k
    break;
1156
7.55k
  case STRESSPOSN_1RH: // LANG=hi stress on the last heaviest syllable
1157
7.55k
    if (stressed_syllable == 0) {
1158
2.87k
      int wt;
1159
2.87k
      int max_weight = -1;
1160
1161
      // find the heaviest syllable, excluding the final syllable
1162
5.48k
      for (ix = 1; ix < (vowel_count-1); ix++) {
1163
2.60k
        if (vowel_stress[ix] < STRESS_IS_DIMINISHED) {
1164
2.44k
          if ((wt = syllable_weight[ix]) >= max_weight) {
1165
2.23k
            max_weight = wt;
1166
2.23k
            stressed_syllable = ix;
1167
2.23k
          }
1168
2.44k
        }
1169
2.60k
      }
1170
1171
2.87k
      if ((syllable_weight[vowel_count-1] == 2) &&  (max_weight < 2)) {
1172
        // the only double=heavy syllable is the final syllable, so stress this
1173
256
        stressed_syllable = vowel_count-1;
1174
2.62k
      } else if (max_weight <= 0) {
1175
        // all syllables, exclusing the last, are light. Stress the first syllable
1176
1.78k
        stressed_syllable = 1;
1177
1.78k
      }
1178
1179
2.87k
      vowel_stress[stressed_syllable] = STRESS_IS_PRIMARY;
1180
2.87k
      max_stress = STRESS_IS_PRIMARY;
1181
2.87k
    }
1182
7.55k
    break;
1183
2.24k
  case STRESSPOSN_1RU : // LANG=tr, the last syllable for any vowel marked explicitly as unstressed
1184
2.24k
    if (stressed_syllable == 0) {
1185
818
      stressed_syllable = vowel_count - 1;
1186
1.44k
      for (ix = 1; ix < vowel_count; ix++) {
1187
634
        if (vowel_stress[ix] == STRESS_IS_UNSTRESSED) {
1188
9
          stressed_syllable = ix-1;
1189
9
          break;
1190
9
        }
1191
634
      }
1192
818
      vowel_stress[stressed_syllable] = STRESS_IS_PRIMARY;
1193
818
      max_stress = STRESS_IS_PRIMARY;
1194
818
    }
1195
2.24k
    break;
1196
0
  case STRESSPOSN_ALL: // mark all as stressed
1197
0
    for (ix = 1; ix < vowel_count; ix++) {
1198
0
      if (vowel_stress[ix] < STRESS_IS_DIMINISHED)
1199
0
        vowel_stress[ix] = STRESS_IS_PRIMARY;
1200
0
    }
1201
0
    break;
1202
14.3k
  case STRESSPOSN_GREENLANDIC: // LANG=kl (Greenlandic)
1203
14.3k
    long_vowel = 0;
1204
31.3k
    for (ix = 1; ix < vowel_count; ix++) {
1205
16.9k
      if (vowel_stress[ix] == STRESS_IS_PRIMARY)
1206
4.72k
        vowel_stress[ix] = STRESS_IS_SECONDARY; // change marked stress (consonant clusters) to secondary (except the last)
1207
1208
16.9k
      if (vowel_length[ix] > 0) {
1209
728
        long_vowel = ix;
1210
728
        vowel_stress[ix] = STRESS_IS_SECONDARY; // give secondary stress to all long vowels
1211
728
      }
1212
16.9k
    }
1213
1214
    // 'stressed_syllable' gives the last marked stress
1215
14.3k
    if (stressed_syllable == 0) {
1216
      // no marked stress, choose the last long vowel
1217
10.5k
      if (long_vowel > 0)
1218
290
        stressed_syllable = long_vowel;
1219
10.2k
      else {
1220
        // no long vowels or consonant clusters
1221
10.2k
        if (vowel_count > 5)
1222
9
          stressed_syllable = vowel_count - 3; // more than 4 syllables
1223
10.2k
        else
1224
10.2k
          stressed_syllable = vowel_count - 1;
1225
10.2k
      }
1226
10.5k
    }
1227
14.3k
    vowel_stress[stressed_syllable] = STRESS_IS_PRIMARY;
1228
14.3k
    max_stress = STRESS_IS_PRIMARY;
1229
14.3k
    break;
1230
10.2k
  case STRESSPOSN_1SL:  // LANG=ml, 1st unless 1st vowel is short and 2nd is long
1231
10.2k
    if (stressed_syllable == 0) {
1232
6.46k
      stressed_syllable = 1;
1233
6.46k
      if ((vowel_length[1] == 0) && (vowel_count > 2) && (vowel_length[2] > 0))
1234
522
        stressed_syllable = 2;
1235
6.46k
      vowel_stress[stressed_syllable] = STRESS_IS_PRIMARY;
1236
6.46k
      max_stress = STRESS_IS_PRIMARY;
1237
6.46k
    }
1238
10.2k
    break;
1239
1240
12.6k
  case STRESSPOSN_EU: // LANG=eu. If more than 2 syllables: primary stress in second syllable and secondary on last.
1241
12.6k
    if ((stressed_syllable == 0) && (vowel_count > 2)) {
1242
2.39k
      for (ix = 1; ix < vowel_count; ix++) {
1243
1.84k
        vowel_stress[ix] = STRESS_IS_DIMINISHED;
1244
1.84k
      }
1245
544
      stressed_syllable = 2;
1246
544
      if (max_stress == STRESS_IS_DIMINISHED)
1247
530
        vowel_stress[stressed_syllable] = STRESS_IS_PRIMARY;
1248
544
      max_stress = STRESS_IS_PRIMARY;
1249
544
      if (vowel_count > 3) {
1250
388
        vowel_stress[vowel_count - 1] = STRESS_IS_SECONDARY;
1251
388
      }
1252
544
    }
1253
12.6k
    break;
1254
858k
  }
1255
1256
858k
  if ((stressflags & S_FINAL_VOWEL_UNSTRESSED) && ((control & 2) == 0) && (vowel_count > 2) && (max_stress_input < STRESS_IS_SECONDARY) && (vowel_stress[vowel_count - 1] == STRESS_IS_PRIMARY)) {
1257
    // Don't allow stress on a word-final vowel
1258
    // Only do this if there is no suffix phonemes to be added, and if a stress position was not given explicitly
1259
1.20k
    if (phoneme_tab[final_ph]->type == phVOWEL) {
1260
902
      vowel_stress[vowel_count - 1] = STRESS_IS_UNSTRESSED;
1261
902
      vowel_stress[vowel_count - 2] = STRESS_IS_PRIMARY;
1262
902
    }
1263
1.20k
  }
1264
1265
  // now guess the complete stress pattern
1266
858k
  if (max_stress < STRESS_IS_PRIMARY)
1267
461k
    stress = STRESS_IS_PRIMARY; // no primary stress marked, use for 1st syllable
1268
396k
  else
1269
396k
    stress = STRESS_IS_SECONDARY;
1270
1271
858k
  if (unstressed_word == false) {
1272
843k
    if ((stressflags & S_2_SYL_2) && (vowel_count == 3)) {
1273
      // Two syllable word, if one syllable has primary stress, then give the other secondary stress
1274
0
      if (vowel_stress[1] == STRESS_IS_PRIMARY)
1275
0
        vowel_stress[2] = STRESS_IS_SECONDARY;
1276
0
      if (vowel_stress[2] == STRESS_IS_PRIMARY)
1277
0
        vowel_stress[1] = STRESS_IS_SECONDARY;
1278
0
    }
1279
1280
843k
    if ((stressflags & S_INITIAL_2) && (vowel_stress[1] < STRESS_IS_DIMINISHED)) {
1281
      // If there is only one syllable before the primary stress, give it a secondary stress
1282
1.18k
      if ((vowel_count > 3) && (vowel_stress[2] >= STRESS_IS_PRIMARY))
1283
501
        vowel_stress[1] = STRESS_IS_SECONDARY;
1284
1.18k
    }
1285
843k
  }
1286
1287
858k
  bool done = false;
1288
858k
  first_primary = 0;
1289
2.62M
  for (v = 1; v < vowel_count; v++) {
1290
1.77M
    if (vowel_stress[v] < STRESS_IS_DIMINISHED) {
1291
1.07M
      if ((stressflags & S_FINAL_NO_2) && (stress < STRESS_IS_PRIMARY) && (v == vowel_count-1)) {
1292
        // flag: don't give secondary stress to final vowel
1293
1.00M
      } else if ((stressflags & 0x8000) && (done == false)) {
1294
58.3k
        vowel_stress[v] = (char)stress;
1295
58.3k
        done = true;
1296
58.3k
        stress = STRESS_IS_SECONDARY; // use secondary stress for remaining syllables
1297
942k
      } else if ((vowel_stress[v-1] <= STRESS_IS_UNSTRESSED) && ((vowel_stress[v+1] <= STRESS_IS_UNSTRESSED) || ((stress == STRESS_IS_PRIMARY) && (vowel_stress[v+1] <= STRESS_IS_NOT_STRESSED)))) {
1298
        // trochaic: give stress to vowel surrounded by unstressed vowels
1299
1300
569k
        if ((stress == STRESS_IS_SECONDARY) && (stressflags & S_NO_AUTO_2))
1301
52.6k
          continue; // don't use secondary stress
1302
1303
        // don't put secondary stress on a light syllable if the rest of the word (excluding last syllable) contains a heavy syllable
1304
516k
        if ((v > 1) && (stressflags & S_2_TO_HEAVY) && (syllable_weight[v] == 0)) {
1305
392
          bool skip = false;
1306
1.31k
          for (int i = v; i < vowel_count - 1; i++) {
1307
981
            if (syllable_weight[i] > 0) {
1308
58
              skip = true;
1309
58
              break;
1310
58
            }
1311
981
          }
1312
392
          if (skip == true)
1313
58
            continue;
1314
392
        }
1315
1316
516k
        if ((v > 1) && (stressflags & S_2_TO_HEAVY) && (syllable_weight[v] == 0) && (syllable_weight[v+1] > 0)) {
1317
          // don't put secondary stress on a light syllable which is followed by a heavy syllable
1318
9
          continue;
1319
9
        }
1320
1321
        // should start with secondary stress on the first syllable, or should it count back from
1322
        // the primary stress and put secondary stress on alternate syllables?
1323
516k
        vowel_stress[v] = (char)stress;
1324
516k
        done = true;
1325
516k
        stress = STRESS_IS_SECONDARY; // use secondary stress for remaining syllables
1326
516k
      }
1327
1.07M
    }
1328
1329
1.71M
    if (vowel_stress[v] >= STRESS_IS_PRIMARY) {
1330
960k
      if (first_primary == 0)
1331
833k
        first_primary = v;
1332
126k
      else if (stressflags & S_FIRST_PRIMARY) {
1333
        // reduce primary stresses after the first to secondary
1334
2.58k
        vowel_stress[v] = STRESS_IS_SECONDARY;
1335
2.58k
      }
1336
960k
    }
1337
1.71M
  }
1338
1339
858k
  if ((unstressed_word) && (tonic < 0)) {
1340
15.5k
    if (vowel_count <= 2)
1341
14.9k
      tonic = tr->langopts.unstressed_wd1; // monosyllable - unstressed
1342
572
    else
1343
572
      tonic = tr->langopts.unstressed_wd2; // more than one syllable, used secondary stress as the main stress
1344
15.5k
  }
1345
1346
858k
  max_stress = STRESS_IS_DIMINISHED;
1347
858k
  max_stress_posn = 0;
1348
2.62M
  for (v = 1; v < vowel_count; v++) {
1349
1.77M
    if (vowel_stress[v] >= max_stress) {
1350
1.02M
      max_stress = vowel_stress[v];
1351
1.02M
      max_stress_posn = v;
1352
1.02M
    }
1353
1.77M
  }
1354
1355
858k
  if (tonic >= 0) {
1356
    // find position of highest stress, and replace it by 'tonic'
1357
1358
    // don't disturb an explicitly set stress by 'unstress-at-end' flag
1359
15.7k
    if ((tonic > max_stress) || (max_stress <= STRESS_IS_PRIMARY))
1360
15.7k
      vowel_stress[max_stress_posn] = (char)tonic;
1361
15.7k
    max_stress = tonic;
1362
15.7k
  }
1363
1364
  // produce output phoneme string
1365
858k
  p = phonetic;
1366
858k
  v = 1;
1367
1368
858k
  if (!(control & 1) && ((ph = phoneme_tab[*p]) != NULL)) {
1369
652k
    while ((ph->type == phSTRESS) || (*p == phonEND_WORD)) {
1370
75.8k
      p++;
1371
75.8k
      ph = phoneme_tab[p[0]];
1372
75.8k
    }
1373
1374
576k
    if ((tr->langopts.vowel_pause & 0x30) && (ph->type == phVOWEL)) {
1375
      // word starts with a vowel
1376
1377
19.8k
      if ((tr->langopts.vowel_pause & 0x20) && (vowel_stress[1] >= STRESS_IS_PRIMARY))
1378
17.4k
        *output++ = phonPAUSE_NOLINK; // not to be replaced by link
1379
2.39k
      else
1380
2.39k
        *output++ = phonPAUSE_VSHORT; // break, but no pause
1381
19.8k
    }
1382
576k
  }
1383
1384
858k
  p = phonetic;
1385
  /* Note: v progression has to strictly follow the vowel_stress production in GetVowelStress */
1386
5.38M
  while (((phcode = *p++) != 0) && (output < max_output)) {
1387
4.52M
    if ((ph = phoneme_tab[phcode]) == NULL)
1388
0
      continue;
1389
1390
4.52M
    if (ph->type == phPAUSE)
1391
356k
      tr->prev_last_stress = 0;
1392
4.17M
    else if (((ph->type == phVOWEL) && !(ph->phflags & phNONSYLLABIC)) || (*p == phonSYLLABIC)) {
1393
      // a vowel, or a consonant followed by a syllabic consonant marker
1394
1395
1.76M
      assert(v <= vowel_count);
1396
1397
1.76M
      v_stress = vowel_stress[v];
1398
1.76M
      tr->prev_last_stress = v_stress;
1399
1400
1.76M
      if (v_stress <= STRESS_IS_UNSTRESSED) {
1401
656k
        if ((v > 1) && (max_stress >= 2) && (stressflags & S_FINAL_DIM) && (v == (vowel_count-1))) {
1402
          // option: mark unstressed final syllable as diminished
1403
102k
          v_stress = STRESS_IS_DIMINISHED;
1404
554k
        } else if ((stressflags & S_NO_DIM) || (v == 1) || (v == (vowel_count-1))) {
1405
          // first or last syllable, or option 'don't set diminished stress'
1406
321k
          v_stress = STRESS_IS_UNSTRESSED;
1407
321k
        } else if ((v == (vowel_count-2)) && (vowel_stress[vowel_count-1] <= STRESS_IS_UNSTRESSED)) {
1408
          // penultimate syllable, followed by an unstressed final syllable
1409
22.0k
          v_stress = STRESS_IS_UNSTRESSED;
1410
210k
        } else {
1411
          // unstressed syllable within a word
1412
210k
          if ((vowel_stress[v-1] < STRESS_IS_DIMINISHED) || ((stressflags & S_MID_DIM) == 0)) {
1413
207k
            v_stress = STRESS_IS_DIMINISHED;
1414
207k
            vowel_stress[v] = v_stress;
1415
207k
          }
1416
210k
        }
1417
656k
      }
1418
1419
1.76M
      if ((v_stress == STRESS_IS_DIMINISHED) || (v_stress > STRESS_IS_UNSTRESSED))
1420
1.41M
        *output++ = stress_phonemes[v_stress]; // mark stress of all vowels except 1 (unstressed)
1421
1422
1.76M
      if (vowel_stress[v] > max_stress)
1423
0
        max_stress = vowel_stress[v];
1424
1425
1.76M
      if ((*p == phonLENGTHEN) && ((opt_length = tr->langopts.param[LOPT_IT_LENGTHEN]) & 1)) {
1426
        // remove lengthen indicator from non-stressed syllables
1427
17.4k
        bool shorten = false;
1428
1429
17.4k
        if (opt_length & 0x10) {
1430
          // only allow lengthen indicator on the highest stress syllable in the word
1431
0
          if (v != max_stress_posn)
1432
0
            shorten = true;
1433
17.4k
        } else if (v_stress < STRESS_IS_PRIMARY) {
1434
          // only allow lengthen indicator if stress >= STRESS_IS_PRIMARY.
1435
2.69k
          shorten = true;
1436
2.69k
        }
1437
1438
17.4k
        if (shorten)
1439
2.69k
          p++;
1440
17.4k
      }
1441
1.76M
      v++;
1442
1.76M
    }
1443
1444
4.52M
    if (phcode != 1)
1445
4.52M
      *output++ = phcode;
1446
4.52M
  }
1447
858k
  *output++ = 0;
1448
1449
858k
  return;
1450
858k
}
1451
1452
void AppendPhonemes(Translator *tr, char *string, int size, const char *ph)
1453
1.00M
{
1454
  /* Add new phoneme string "ph" to "string"
1455
      Keeps count of the number of vowel phonemes in the word, and whether these
1456
     can be stressed syllables.  These values can be used in translation rules
1457
   */
1458
1459
1.00M
  const char *p;
1460
1.00M
  unsigned char c;
1461
1.00M
  int length;
1462
1463
1.00M
  length = strlen(ph) + strlen(string);
1464
1.00M
  if (length >= size)
1465
93.2k
    return;
1466
1467
  // any stressable vowel ?
1468
1.00M
  bool unstress_mark = false;
1469
913k
  p = ph;
1470
2.54M
  while ((c = *p++) != 0) {
1471
1.63M
    if (c >= n_phoneme_tab) continue;
1472
1473
1.62M
    if (!phoneme_tab[c]) continue;
1474
1475
1.62M
    if (phoneme_tab[c]->type == phSTRESS) {
1476
113k
      if (phoneme_tab[c]->std_length < 4)
1477
36.3k
        unstress_mark = true;
1478
1.51M
    } else {
1479
1.51M
      if (phoneme_tab[c]->type == phVOWEL) {
1480
681k
        if (((phoneme_tab[c]->phflags & phUNSTRESSED) == 0) &&
1481
621k
            (unstress_mark == false)) {
1482
603k
          tr->word_stressed_count++;
1483
603k
        }
1484
681k
        unstress_mark = false;
1485
681k
        tr->word_vowel_count++;
1486
681k
      }
1487
1.51M
    }
1488
1.62M
  }
1489
1490
913k
  if (string != NULL)
1491
913k
    strcat(string, ph);
1492
913k
}
1493
1494
static void MatchRule(Translator *tr, char *word[], char *word_start, int group_length, char *rule, MatchRecord *match_out, int word_flags, int dict_flags)
1495
2.30M
{
1496
  /* Checks a specified word against dictionary rules.
1497
      Returns with phoneme code string, or NULL if no match found.
1498
1499
      word (indirect) points to current character group within the input word
1500
              This is advanced by this procedure as characters are consumed
1501
1502
      group:  the initial characters used to choose the rules group
1503
1504
      rule:  address of dictionary rule data for this character group
1505
1506
      match_out:  returns best points score
1507
1508
      word_flags:  indicates whether this is a retranslation after a suffix has been removed
1509
   */
1510
1511
2.30M
  unsigned char rb;     // current instuction from rule
1512
2.30M
  unsigned char letter; // current letter from input word, single byte
1513
2.30M
  int letter_w;         // current letter, wide character
1514
2.30M
  int last_letter_w;    // last letter, wide character
1515
2.30M
  int letter_xbytes;    // number of extra bytes of multibyte character (num bytes - 1)
1516
1517
2.30M
  char *pre_ptr;
1518
2.30M
  char *post_ptr;       // pointer to first character after group
1519
1520
2.30M
  char *rule_start;     // start of current match template
1521
2.30M
  char *p;
1522
2.30M
  int match_type;       // left, right, or consume
1523
2.30M
  int syllable_count;
1524
2.30M
  int vowel;
1525
2.30M
  int letter_group;
1526
2.30M
  int lg_pts;
1527
2.30M
  int n_bytes;
1528
2.30M
  int add_points;
1529
2.30M
  int command;
1530
1531
2.30M
  MatchRecord match;
1532
2.30M
  MatchRecord best;
1533
1534
2.30M
  int total_consumed; // letters consumed for best match
1535
1536
2.30M
  unsigned char condition_num;
1537
2.30M
  char *common_phonemes; // common to a group of entries
1538
2.30M
  char *group_chars;
1539
2.30M
  char word_buf[N_WORD_BYTES];
1540
1541
2.30M
  group_chars = *word;
1542
1543
2.30M
  if (rule == NULL) {
1544
96.2k
    match_out->points = 0;
1545
96.2k
    (*word)++;
1546
96.2k
    return;
1547
96.2k
  }
1548
1549
2.20M
  total_consumed = 0;
1550
2.20M
  common_phonemes = NULL;
1551
1552
2.20M
  best.points = 0;
1553
2.20M
  best.phonemes = "";
1554
2.20M
  best.end_type = 0;
1555
2.20M
  best.del_fwd = NULL;
1556
1557
  // search through dictionary rules
1558
96.0M
  while (rule[0] != RULE_GROUP_END) {
1559
93.8M
    bool check_atstart = false;
1560
93.8M
    int consumed = 0;         // number of letters consumed from input
1561
93.8M
    int distance_left = -2;
1562
93.8M
        int distance_right = -6; // used to reduce points for matches further away the current letter
1563
93.8M
    int failed = 0;
1564
93.8M
    int unpron_ignore = word_flags & FLAG_UNPRON_TEST;
1565
1566
93.8M
    match_type = 0;
1567
93.8M
    letter_w = 0;
1568
1569
93.8M
    match.points = 1;
1570
93.8M
    match.end_type = 0;
1571
93.8M
    match.del_fwd = NULL;
1572
1573
93.8M
    pre_ptr = *word;
1574
93.8M
    post_ptr = *word + group_length;
1575
1576
    // work through next rule until end, or until no-match proved
1577
93.8M
    rule_start = rule;
1578
1579
258M
    while (!failed) {
1580
165M
      rb = *rule++;
1581
165M
      add_points = 0;
1582
1583
165M
      if (rb <= RULE_LINENUM) {
1584
63.5M
        switch (rb)
1585
63.5M
        {
1586
314k
        case 0: // no phoneme string for this rule, use previous common rule
1587
314k
          if (common_phonemes != NULL) {
1588
314k
            match.phonemes = common_phonemes;
1589
1.58M
            while (((rb = *match.phonemes++) != 0) && (rb != RULE_PHONEMES)) {
1590
1.27M
              if (rb == RULE_CONDITION)
1591
817
                match.phonemes++; // skip over condition number
1592
1.27M
              if (rb == RULE_LINENUM)
1593
0
                match.phonemes += 2; // skip over line number
1594
1.27M
            }
1595
314k
          } else
1596
0
            match.phonemes = "";
1597
314k
          rule--; // so we are still pointing at the 0
1598
314k
          failed = 2; // matched OK
1599
314k
          break;
1600
8.00M
        case RULE_PRE_ATSTART: // pre rule with implied 'start of word'
1601
8.00M
          check_atstart = true;
1602
8.00M
          unpron_ignore = 0;
1603
8.00M
          match_type = RULE_PRE;
1604
8.00M
          break;
1605
23.8M
        case RULE_PRE:
1606
23.8M
          match_type = RULE_PRE;
1607
23.8M
          if (word_flags & FLAG_UNPRON_TEST) {
1608
            // checking the start of the word for unpronouncable character sequences, only
1609
            // consider rules which explicitly match the start of a word
1610
            // Note: Those rules now use RULE_PRE_ATSTART
1611
395k
            failed = 1;
1612
395k
          }
1613
23.8M
          break;
1614
17.8M
        case RULE_POST:
1615
17.8M
          match_type = RULE_POST;
1616
17.8M
          break;
1617
1.21M
        case RULE_PHONEMES:
1618
1.21M
          match.phonemes = rule;
1619
1.21M
          failed = 2; // matched OK
1620
1.21M
          break;
1621
11.8M
        case RULE_PH_COMMON:
1622
11.8M
          common_phonemes = rule;
1623
11.8M
          break;
1624
494k
        case RULE_CONDITION:
1625
          // conditional rule, next byte gives condition number
1626
494k
          condition_num = *rule++;
1627
1628
494k
          if (condition_num >= 32) {
1629
            // allow the rule only if the condition number is NOT set
1630
65.9k
            if ((tr->dict_condition & (1L << (condition_num-32))) != 0)
1631
74
              failed = 1;
1632
428k
          } else {
1633
            // allow the rule only if the condition number is set
1634
428k
            if ((tr->dict_condition & (1L << condition_num)) == 0)
1635
353k
              failed = 1;
1636
428k
          }
1637
1638
494k
          if (!failed)
1639
140k
            match.points++; // add one point for a matched conditional rule
1640
494k
          break;
1641
0
        case RULE_LINENUM:
1642
0
          rule += 2;
1643
0
          break;
1644
63.5M
        }
1645
63.5M
        continue;
1646
63.5M
      }
1647
1648
101M
      switch (match_type)
1649
101M
      {
1650
53.6M
      case 0:
1651
        // match and consume this letter
1652
53.6M
        letter = *post_ptr++;
1653
1654
53.6M
        if ((letter == rb) || ((letter == (unsigned char)REPLACED_E) && (rb == 'e'))) {
1655
2.33M
          if ((letter & 0xc0) != 0x80)
1656
2.15M
            add_points = 21; // don't add point for non-initial UTF-8 bytes
1657
2.33M
          consumed++;
1658
2.33M
        } else
1659
51.3M
          failed = 1;
1660
53.6M
        break;
1661
19.8M
      case RULE_POST:
1662
        // continue moving forwards
1663
19.8M
        distance_right += 6;
1664
19.8M
        if (distance_right > 18)
1665
18.7k
          distance_right = 19;
1666
19.8M
        last_letter_w = letter_w;
1667
19.8M
        if (!post_ptr[-1]) {
1668
          // we had already reached the end of text!
1669
          // reading after that does not make sense, that cannot match
1670
17
          failed = 1;
1671
17
          break;
1672
17
        }
1673
19.8M
        letter_xbytes = utf8_in(&letter_w, post_ptr)-1;
1674
19.8M
        letter = *post_ptr++;
1675
1676
19.8M
        switch (rb)
1677
19.8M
        {
1678
3.88M
        case RULE_LETTERGP:
1679
3.88M
          letter_group = LetterGroupNo(rule++);
1680
3.88M
          if (IsLetter(tr, letter_w, letter_group)) {
1681
1.07M
            lg_pts = 20;
1682
1.07M
            if (letter_group == 2)
1683
619k
              lg_pts = 19; // fewer points for C, general consonant
1684
1.07M
            add_points = (lg_pts-distance_right);
1685
1.07M
            post_ptr += letter_xbytes;
1686
1.07M
          } else
1687
2.81M
            failed = 1;
1688
3.88M
          break;
1689
457k
        case RULE_LETTERGP2: // match against a list of utf-8 strings
1690
457k
          letter_group = LetterGroupNo(rule++);
1691
457k
          if ((n_bytes = IsLetterGroup(tr, post_ptr-1, letter_group, 0)) >= 0) {
1692
136k
            add_points = (20-distance_right);
1693
            // move pointer, if group was found
1694
136k
            post_ptr += (n_bytes-1);
1695
136k
          } else
1696
320k
            failed = 1;
1697
457k
          break;
1698
11.7k
        case RULE_NOTVOWEL:
1699
11.7k
          if (IsLetter(tr, letter_w, 0) || ((letter_w == ' ') && (word_flags & FLAG_SUFFIX_VOWEL)))
1700
5.14k
            failed = 1;
1701
6.63k
          else {
1702
6.63k
            add_points = (20-distance_right);
1703
6.63k
            post_ptr += letter_xbytes;
1704
6.63k
          }
1705
11.7k
          break;
1706
12.1k
        case RULE_DIGIT:
1707
12.1k
          if (IsDigit(letter_w)) {
1708
3.89k
            add_points = (20-distance_right);
1709
3.89k
            post_ptr += letter_xbytes;
1710
8.26k
          } else if (tr->langopts.tone_numbers) {
1711
            // also match if there is no digit
1712
36
            add_points = (20-distance_right);
1713
36
            post_ptr--;
1714
36
          } else
1715
8.22k
            failed = 1;
1716
12.1k
          break;
1717
0
        case RULE_NONALPHA:
1718
0
          if (!iswalpha(letter_w)) {
1719
0
            add_points = (21-distance_right);
1720
0
            post_ptr += letter_xbytes;
1721
0
          } else
1722
0
            failed = 1;
1723
0
          break;
1724
12.0k
        case RULE_DOUBLE:
1725
12.0k
          if (letter_w == last_letter_w) {
1726
897
            add_points = (21-distance_right);
1727
897
            post_ptr += letter_xbytes;
1728
897
          } else
1729
11.1k
            failed = 1;
1730
12.0k
          break;
1731
167k
        case RULE_DOLLAR:
1732
167k
          post_ptr--;
1733
167k
          command = *rule++;
1734
167k
          if (command == DOLLAR_UNPR)
1735
54.2k
            match.end_type = SUFX_UNPRON; // $unpron
1736
113k
          else if (command == DOLLAR_NOPREFIX) { // $noprefix
1737
253
            if (word_flags & FLAG_PREFIX_REMOVED)
1738
0
              failed = 1; // a prefix has been removed
1739
253
            else
1740
253
              add_points = 1;
1741
113k
          } else if ((command & 0xf0) == 0x10) {
1742
            // $w_alt
1743
112k
            if (dict_flags & (1 << (BITNUM_FLAG_ALT + (command & 0xf))))
1744
215
              add_points = 23;
1745
112k
            else
1746
112k
              failed = 1;
1747
112k
          } else if (((command & 0xf0) == 0x20) || (command == DOLLAR_LIST)) {
1748
674
            DollarRule(word, word_start, consumed, group_length, word_buf, tr, command, &failed, &add_points);
1749
674
          }
1750
1751
167k
          break;
1752
127k
        case '-':
1753
127k
          if ((letter == '-') || ((letter == ' ') && (word_flags & FLAG_HYPHEN_AFTER)))
1754
445
            add_points = (22-distance_right); // one point more than match against space
1755
126k
          else
1756
126k
            failed = 1;
1757
127k
          break;
1758
36.0k
        case RULE_SYLLABLE:
1759
36.0k
        {
1760
          // more than specified number of vowel letters to the right
1761
36.0k
          char *p = post_ptr + letter_xbytes;
1762
36.0k
          int vowel_count = 0;
1763
1764
36.0k
          syllable_count = 1;
1765
40.6k
          while (*rule == RULE_SYLLABLE) {
1766
4.56k
            rule++;
1767
4.56k
            syllable_count += 1; // number of syllables to match
1768
4.56k
          }
1769
36.0k
          vowel = 0;
1770
166k
          while (letter_w != RULE_SPACE && letter_w != 0) {
1771
130k
            if ((vowel == 0) && IsLetter(tr, letter_w, LETTERGP_VOWEL2)) {
1772
              // this is counting vowels which are separated by non-vowel letters
1773
44.6k
              vowel_count++;
1774
44.6k
            }
1775
130k
            vowel = IsLetter(tr, letter_w, LETTERGP_VOWEL2);
1776
130k
            p += utf8_in(&letter_w, p);
1777
130k
          }
1778
36.0k
          if (syllable_count <= vowel_count)
1779
23.3k
            add_points = (18+syllable_count-distance_right);
1780
12.7k
          else
1781
12.7k
            failed = 1;
1782
36.0k
        }
1783
36.0k
          break;
1784
47.4k
        case RULE_NOVOWELS:
1785
47.4k
        {
1786
47.4k
          char *p = post_ptr + letter_xbytes;
1787
66.9k
          while (letter_w != RULE_SPACE && letter_w != 0) {
1788
59.6k
            if (IsLetter(tr, letter_w, LETTERGP_VOWEL2)) {
1789
40.0k
              failed = 1;
1790
40.0k
              break;
1791
40.0k
            }
1792
19.5k
            p += utf8_in(&letter_w, p);
1793
19.5k
          }
1794
47.4k
          if (!failed)
1795
7.37k
            add_points = (19-distance_right);
1796
47.4k
        }
1797
47.4k
          break;
1798
40.7k
        case RULE_SKIPCHARS:
1799
40.7k
        {
1800
          // '(Jxy'  means 'skip characters until xy'
1801
40.7k
          char *p = post_ptr - 1; // to allow empty jump (without letter between), go one back
1802
40.7k
          char *p2 = p;   // pointer to the previous character in the word
1803
40.7k
          int rule_w;   // first wide character of skip rule
1804
40.7k
          utf8_in(&rule_w, rule);
1805
40.7k
          int g_bytes = -1; // bytes of successfully found character group
1806
7.51M
          while ((letter_w != rule_w) && (letter_w != RULE_SPACE) && (letter_w != 0) && (g_bytes == -1)) {
1807
7.47M
            if (rule_w == RULE_LETTERGP2)
1808
4.92M
              g_bytes = IsLetterGroup(tr, p, LetterGroupNo(rule + 1), 0);
1809
7.47M
            p2 = p;
1810
7.47M
            p += utf8_in(&letter_w, p);
1811
7.47M
          }
1812
40.7k
          if ((letter_w == rule_w) || (g_bytes >= 0))
1813
13.1k
            post_ptr = p2;
1814
40.7k
        }
1815
40.7k
          break;
1816
10.3k
        case RULE_INC_SCORE:
1817
10.3k
          post_ptr--;
1818
10.3k
          add_points = 20; // force an increase in points
1819
10.3k
          break;
1820
8.11k
        case RULE_DEC_SCORE:
1821
8.11k
          post_ptr--;
1822
8.11k
          add_points = -20; // force an decrease in points
1823
8.11k
          break;
1824
5.41k
        case RULE_DEL_FWD:
1825
          // find the next 'e' in the word and replace by 'E'
1826
10.7k
          for (p = *word + group_length; p < post_ptr; p++) {
1827
10.7k
            if (*p == 'e') {
1828
5.40k
              match.del_fwd = p;
1829
5.40k
              break;
1830
5.40k
            }
1831
10.7k
          }
1832
5.41k
          break;
1833
31.3k
        case RULE_ENDING:
1834
31.3k
        {
1835
31.3k
          int end_type;
1836
          // next 3 bytes are a (non-zero) ending type. 2 bytes of flags + suffix length
1837
31.3k
          end_type = (rule[0] << 16) + ((rule[1] & 0x7f) << 8) + (rule[2] & 0x7f);
1838
1839
31.3k
          if ((tr->word_vowel_count == 0) && !(end_type & SUFX_P) && (tr->langopts.param[LOPT_SUFFIX] & 1))
1840
655
            failed = 1; // don't match a suffix rule if there are no previous syllables (needed for lang=tr).
1841
30.6k
          else {
1842
30.6k
            match.end_type = end_type;
1843
30.6k
            rule += 3;
1844
30.6k
          }
1845
31.3k
        }
1846
31.3k
          break;
1847
11.3k
        case RULE_NO_SUFFIX:
1848
11.3k
          if (word_flags & FLAG_SUFFIX_REMOVED)
1849
94
            failed = 1; // a suffix has been removed
1850
11.2k
          else {
1851
11.2k
            post_ptr--;
1852
11.2k
            add_points = 1;
1853
11.2k
          }
1854
11.3k
          break;
1855
15.0M
        default:
1856
15.0M
          if (letter == rb) {
1857
1.00M
            if ((letter & 0xc0) != 0x80) {
1858
              // not for non-initial UTF-8 bytes
1859
921k
              add_points = (21-distance_right);
1860
921k
            }
1861
1.00M
          } else
1862
14.0M
            failed = 1;
1863
15.0M
          break;
1864
19.8M
        }
1865
19.8M
        break;
1866
27.9M
      case RULE_PRE:
1867
        // match backwards from start of current group
1868
27.9M
        distance_left += 2;
1869
27.9M
        if (distance_left > 18)
1870
125
          distance_left = 19;
1871
1872
27.9M
        if (!*pre_ptr) {
1873
          // we had already reached the beginning of text!
1874
          // reading before this does not make sense, that cannot match
1875
0
          failed = 1;
1876
0
          break;
1877
0
        }
1878
27.9M
        utf8_in(&last_letter_w, pre_ptr);
1879
27.9M
        pre_ptr--;
1880
27.9M
        letter_xbytes = utf8_in2(&letter_w, pre_ptr, 1)-1;
1881
27.9M
        letter = *pre_ptr;
1882
1883
27.9M
        switch (rb)
1884
27.9M
        {
1885
2.08M
        case RULE_LETTERGP:
1886
2.08M
          letter_group = LetterGroupNo(rule++);
1887
2.08M
          if (IsLetter(tr, letter_w, letter_group)) {
1888
549k
            lg_pts = 20;
1889
549k
            if (letter_group == 2)
1890
452k
              lg_pts = 19; // fewer points for C, general consonant
1891
549k
            add_points = (lg_pts-distance_left);
1892
549k
            pre_ptr -= letter_xbytes;
1893
549k
          } else
1894
1.53M
            failed = 1;
1895
2.08M
          break;
1896
161k
        case RULE_LETTERGP2: // match against a list of utf-8 strings
1897
161k
          letter_group = LetterGroupNo(rule++);
1898
161k
          if ((n_bytes = IsLetterGroup(tr, pre_ptr, letter_group, 1)) >= 0) {
1899
79.3k
            add_points = (20-distance_right);
1900
            // move pointer, if group was found
1901
79.3k
            pre_ptr -= (n_bytes-1);
1902
79.3k
          } else
1903
82.0k
            failed = 1;
1904
161k
          break;
1905
32.4k
        case RULE_NOTVOWEL:
1906
32.4k
          if (!IsLetter(tr, letter_w, 0)) {
1907
14.9k
            add_points = (20-distance_left);
1908
14.9k
            pre_ptr -= letter_xbytes;
1909
14.9k
          } else
1910
17.4k
            failed = 1;
1911
32.4k
          break;
1912
41.7k
        case RULE_DOUBLE:
1913
41.7k
          if (letter_w == last_letter_w) {
1914
2.96k
            add_points = (21-distance_left);
1915
2.96k
            pre_ptr -= letter_xbytes;
1916
2.96k
          } else
1917
38.7k
            failed = 1;
1918
41.7k
          break;
1919
18.5k
        case RULE_DIGIT:
1920
18.5k
          if (IsDigit(letter_w)) {
1921
1.66k
            add_points = (21-distance_left);
1922
1.66k
            pre_ptr -= letter_xbytes;
1923
1.66k
          } else
1924
16.8k
            failed = 1;
1925
18.5k
          break;
1926
0
        case RULE_NONALPHA:
1927
0
          if (!iswalpha(letter_w)) {
1928
0
            add_points = (21-distance_right);
1929
0
            pre_ptr -= letter_xbytes;
1930
0
          } else
1931
0
            failed = 1;
1932
0
          break;
1933
0
        case RULE_DOLLAR:
1934
0
          pre_ptr++;
1935
0
          command = *rule++;
1936
0
          if ((command == DOLLAR_LIST) || ((command & 0xf0) == 0x20)) {
1937
0
            DollarRule(word, word_start, consumed, group_length, word_buf, tr, command, &failed, &add_points);
1938
0
          }
1939
0
          break;
1940
2.26M
        case RULE_SYLLABLE:
1941
          // more than specified number of vowels to the left
1942
2.26M
          syllable_count = 1;
1943
2.33M
          while (*rule == RULE_SYLLABLE) {
1944
62.3k
            rule++;
1945
62.3k
            syllable_count++; // number of syllables to match
1946
62.3k
          }
1947
2.26M
          if (syllable_count <= tr->word_vowel_count)
1948
1.80M
            add_points = (18+syllable_count-distance_left);
1949
466k
          else
1950
466k
            failed = 1;
1951
2.26M
          break;
1952
1.82M
        case RULE_STRESSED:
1953
1.82M
          pre_ptr++;
1954
1.82M
          if (tr->word_stressed_count > 0)
1955
1.39M
            add_points = 19;
1956
427k
          else
1957
427k
            failed = 1;
1958
1.82M
          break;
1959
581k
        case RULE_NOVOWELS:
1960
581k
        {
1961
581k
          char *p = pre_ptr - letter_xbytes;
1962
998k
          while (letter_w != RULE_SPACE) {
1963
807k
            if (IsLetter(tr, letter_w, LETTERGP_VOWEL2)) {
1964
390k
              failed = 1;
1965
390k
              break;
1966
390k
            }
1967
417k
            p -= utf8_in2(&letter_w, p-1, 1);
1968
417k
          }
1969
581k
          if (!failed)
1970
190k
            add_points = 3;
1971
581k
        }
1972
581k
          break;
1973
1.25k
        case RULE_IFVERB:
1974
1.25k
          pre_ptr++;
1975
1.25k
          if (tr->expect_verb)
1976
190
            add_points = 1;
1977
1.06k
          else
1978
1.06k
            failed = 1;
1979
1.25k
          break;
1980
31
        case RULE_CAPITAL:
1981
31
          pre_ptr++;
1982
31
          if (word_flags & FLAG_FIRST_UPPER)
1983
21
            add_points = 1;
1984
10
          else
1985
10
            failed = 1;
1986
31
          break;
1987
7.96k
        case '.':
1988
          // dot in pre- section, match on any dot before this point in the word
1989
16.7k
          for (p = pre_ptr; *p && *p != ' '; p--) {
1990
9.05k
            if (*p == '.') {
1991
218
              add_points = 50;
1992
218
              break;
1993
218
            }
1994
9.05k
          }
1995
7.96k
          if (!*p || *p == ' ')
1996
7.74k
            failed = 1;
1997
7.96k
          break;
1998
92.4k
        case '-':
1999
92.4k
          if ((letter == '-') || ((letter == ' ') && (word_flags & FLAG_HYPHEN)))
2000
503
            add_points = (22-distance_right); // one point more than match against space
2001
91.9k
          else
2002
91.9k
            failed = 1;
2003
92.4k
          break;
2004
2005
11.6k
        case RULE_SKIPCHARS: {
2006
          // 'xyJ)'  means 'skip characters backwards until xy'
2007
11.6k
          char *p = pre_ptr + 1;  // to allow empty jump (without letter between), go one forward
2008
11.6k
          char *p2 = p;   // pointer to previous character in word
2009
11.6k
          int g_bytes = -1; // bytes of successfully found character group
2010
2011
2.53M
          while ((*p != *rule) && (*p != RULE_SPACE) && (*p != 0) && (g_bytes == -1)) {
2012
2.52M
            p2 = p;
2013
2.52M
            p--;
2014
2.52M
            if (*rule == RULE_LETTERGP2)
2015
2.52M
              g_bytes = IsLetterGroup(tr, p2, LetterGroupNo(rule + 1), 1);
2016
2.52M
          }
2017
2018
          // if succeed, set pre_ptr to next character after 'xy' and remaining
2019
          // 'xy' part is checked as usual in following cycles of PRE rule characters
2020
11.6k
          if (*p == *rule)
2021
234
            pre_ptr = p2;
2022
11.6k
          if (g_bytes >= 0)
2023
1.01k
            pre_ptr = p2 + 1;
2024
2025
11.6k
        }
2026
11.6k
          break;
2027
2028
20.8M
        default:
2029
20.8M
          if (letter == rb) {
2030
1.08M
            if (letter == RULE_SPACE)
2031
59.8k
              add_points = 4;
2032
1.02M
            else if ((letter & 0xc0) != 0x80) {
2033
              // not for non-initial UTF-8 bytes
2034
937k
              add_points = (21-distance_left);
2035
937k
            }
2036
1.08M
          } else
2037
19.7M
            failed = 1;
2038
20.8M
          break;
2039
27.9M
        }
2040
27.9M
        break;
2041
101M
      }
2042
2043
101M
      if (failed == 0)
2044
9.89M
        match.points += add_points;
2045
101M
    }
2046
2047
93.8M
    if ((failed == 2) && (unpron_ignore == 0)) {
2048
      // do we also need to check for 'start of word' ?
2049
1.44M
      if ((check_atstart == false) || (pre_ptr[-1] == ' ')) {
2050
1.29M
        if (check_atstart)
2051
73.0k
          match.points += 4;
2052
2053
        // matched OK, is this better than the last best match ?
2054
1.29M
        if (match.points >= best.points) {
2055
1.10M
          memcpy(&best, &match, sizeof(match));
2056
1.10M
          total_consumed = consumed;
2057
1.10M
        }
2058
2059
1.29M
        if ((option_phonemes & espeakPHONEMES_TRACE) && (match.points > 0) && ((word_flags & FLAG_NO_TRACE) == 0)) {
2060
          // show each rule that matches, and it's points score
2061
0
          int pts;
2062
0
          char decoded_phonemes[80];
2063
0
          char output[80];
2064
2065
0
          pts = match.points;
2066
0
          if (group_length > 1)
2067
0
            pts += 35; // to account for an extra letter matching
2068
0
          DecodePhonemes(match.phonemes, decoded_phonemes);
2069
0
          fprintf(f_trans, "%3d\t%s [%s]\n", pts, DecodeRule(group_chars, group_length, rule_start, word_flags, output), decoded_phonemes);
2070
0
        }
2071
1.29M
      }
2072
1.44M
    }
2073
2074
    // skip phoneme string to reach start of next template
2075
608M
    while (*rule++ != 0) ;
2076
93.8M
  }
2077
2078
  // advance input data pointer
2079
2.20M
  total_consumed += group_length;
2080
2.20M
  if (total_consumed == 0)
2081
1.17M
    total_consumed = 1; // always advance over 1st letter
2082
2083
2.20M
  *word += total_consumed;
2084
2085
2.20M
  if (best.points == 0)
2086
1.28M
    best.phonemes = "";
2087
2.20M
  memcpy(match_out, &best, sizeof(MatchRecord));
2088
2.20M
}
2089
2090
int TranslateRules(Translator *tr, char *p_start, char *phonemes, int ph_size, char *end_phonemes, int word_flags, unsigned int *dict_flags)
2091
1.00M
{
2092
  /* Translate a word bounded by space characters
2093
     Append the result to 'phonemes' and any standard prefix/suffix in 'end_phonemes' */
2094
2095
1.00M
  unsigned char c, c2;
2096
1.00M
  unsigned int c12;
2097
1.00M
  int wc = 0;
2098
1.00M
  char *p2;           // copy of p for use in double letter chain match
2099
1.00M
  int found;
2100
1.00M
  int g;              // group chain number
2101
1.00M
  int g1;             // first group for this letter
2102
1.00M
  int letter;
2103
1.00M
  int any_alpha = 0;
2104
1.00M
  int ix;
2105
1.00M
  unsigned int digit_count = 0;
2106
1.00M
  char *p;
2107
1.00M
  char word_buf[5];
2108
1.00M
  const ALPHABET *alphabet;
2109
1.00M
  int dict_flags0 = 0;
2110
1.00M
  MatchRecord match1 = { 0 };
2111
1.00M
  MatchRecord match2 = { 0 };
2112
1.00M
  char ph_buf[N_PHONEME_BYTES];
2113
1.00M
  char word_copy[N_WORD_BYTES];
2114
1.00M
  static const char str_pause[2] = { phonPAUSE_NOLINK, 0 };
2115
2116
1.00M
  if (tr->data_dictrules == NULL)
2117
0
    return 0;
2118
2119
1.00M
  if (dict_flags != NULL)
2120
651k
    dict_flags0 = dict_flags[0];
2121
2122
60.6M
  for (ix = 0; ix < (N_WORD_BYTES-1);) {
2123
60.3M
    c = p_start[ix];
2124
60.3M
    word_copy[ix++] = c;
2125
60.3M
    if (c == 0)
2126
775k
      break;
2127
60.3M
  }
2128
1.00M
  word_copy[ix] = 0;
2129
2130
1.00M
  if ((option_phonemes & espeakPHONEMES_TRACE) && ((word_flags & FLAG_NO_TRACE) == 0)) {
2131
0
    char wordbuf[120];
2132
0
    unsigned int ix;
2133
2134
0
    for (ix = 0; ((c = p_start[ix]) != ' ') && (c != 0) && (ix < (sizeof(wordbuf)-1)); ix++)
2135
0
      wordbuf[ix] = c;
2136
0
    wordbuf[ix] = 0;
2137
0
    if (word_flags & FLAG_UNPRON_TEST)
2138
0
      fprintf(f_trans, "Unpronouncable? '%s'\n", wordbuf);
2139
0
    else
2140
0
      fprintf(f_trans, "Translate '%s'\n", wordbuf);
2141
0
  }
2142
2143
1.00M
  p = p_start;
2144
1.00M
  tr->word_vowel_count = 0;
2145
1.00M
  tr->word_stressed_count = 0;
2146
2147
1.00M
  if (end_phonemes != NULL)
2148
644k
    end_phonemes[0] = 0;
2149
2150
3.04M
  while (((c = *p) != ' ') && (c != 0)) {
2151
2.28M
    int wc_bytes = utf8_in(&wc, p);
2152
2.28M
    if (IsAlpha(wc))
2153
1.26M
      any_alpha++;
2154
2155
2.28M
    int n = tr->groups2_count[c];
2156
2.28M
    if (IsDigit(wc) && ((tr->langopts.tone_numbers == 0) || !any_alpha)) {
2157
      // lookup the number in *_list not *_rules
2158
92.1k
      char string[8];
2159
92.1k
      char buf[40];
2160
92.1k
      string[0] = '_';
2161
92.1k
      memcpy(&string[1], p, wc_bytes);
2162
92.1k
      string[1+wc_bytes] = 0;
2163
92.1k
      Lookup(tr, string, buf);
2164
92.1k
      if (++digit_count >= 2) {
2165
40.4k
        strcat(buf, str_pause);
2166
40.4k
        digit_count = 0;
2167
40.4k
      }
2168
92.1k
      AppendPhonemes(tr, phonemes, ph_size, buf);
2169
92.1k
      p += wc_bytes;
2170
92.1k
      continue;
2171
2.19M
    } else {
2172
2.19M
      digit_count = 0;
2173
2.19M
      found = 0;
2174
2175
2.19M
      if (((ix = wc - tr->letter_bits_offset) >= 0) && (ix < 128)) {
2176
1.35M
        if (tr->groups3[ix] != NULL) {
2177
68.9k
          MatchRule(tr, &p, p_start, wc_bytes, tr->groups3[ix], &match1, word_flags, dict_flags0);
2178
68.9k
          found = 1;
2179
68.9k
        }
2180
1.35M
      }
2181
2182
2.19M
      if (!found && (n > 0)) {
2183
        // there are some 2 byte chains for this initial letter
2184
413k
        c2 = p[1];
2185
413k
        c12 = c + (c2 << 8); // 2 characters
2186
2187
413k
        g1 = tr->groups2_start[c];
2188
3.44M
        for (g = g1; g < (g1+n); g++) {
2189
3.02M
          if (tr->groups2_name[g] == c12) {
2190
104k
            found = 1;
2191
2192
104k
            p2 = p;
2193
104k
            MatchRule(tr, &p2, p_start, 2, tr->groups2[g], &match2, word_flags, dict_flags0);
2194
104k
            if (match2.points > 0)
2195
69.6k
              match2.points += 35; // to acount for 2 letters matching
2196
2197
            // now see whether single letter chain gives a better match ?
2198
104k
            MatchRule(tr, &p, p_start, 1, tr->groups1[c], &match1, word_flags, dict_flags0);
2199
2200
104k
            if (match2.points >= match1.points) {
2201
              // use match from the 2-letter group
2202
72.6k
              memcpy(&match1, &match2, sizeof(MatchRecord));
2203
72.6k
              p = p2;
2204
72.6k
            }
2205
104k
          }
2206
3.02M
        }
2207
413k
      }
2208
2209
2.19M
      if (!found) {
2210
        // alphabetic, single letter chain
2211
2.02M
        if (tr->groups1[c] != NULL)
2212
748k
          MatchRule(tr, &p, p_start, 1, tr->groups1[c], &match1, word_flags, dict_flags0);
2213
1.27M
        else {
2214
          // no group for this letter, use default group
2215
1.27M
          MatchRule(tr, &p, p_start, 0, tr->groups1[0], &match1, word_flags, dict_flags0);
2216
2217
1.27M
          if ((match1.points == 0) && ((option_sayas & 0x10) == 0)) {
2218
1.23M
            n = utf8_in(&letter, p-1)-1;
2219
2220
1.23M
            if (tr->letter_bits_offset > 0) {
2221
              // not a Latin alphabet, switch to the default Latin alphabet language
2222
523k
              if ((letter <= 0x241) && iswalpha(letter)) {
2223
80.9k
                sprintf(phonemes, "%cen", phonSWITCH);
2224
80.9k
                return 0;
2225
80.9k
              }
2226
523k
            }
2227
2228
            // is it a bracket ?
2229
1.15M
            if (letter == 0xe000+'(') {
2230
5.23k
              if (pre_pause < tr->langopts.param[LOPT_BRACKET_PAUSE_ANNOUNCED])
2231
4.01k
                pre_pause = tr->langopts.param[LOPT_BRACKET_PAUSE_ANNOUNCED]; // a bracket, already spoken by AnnouncePunctuation()
2232
5.23k
            }
2233
1.15M
            if (IsBracket(letter)) {
2234
243k
              if (pre_pause < tr->langopts.param[LOPT_BRACKET_PAUSE])
2235
71.9k
                pre_pause = tr->langopts.param[LOPT_BRACKET_PAUSE];
2236
243k
            }
2237
2238
            // no match, try removing the accent and re-translating the word
2239
1.15M
            if ((letter >= 0xc0) && (letter < N_REMOVE_ACCENT) && ((ix = remove_accent[letter-0xc0]) != 0)) {
2240
              // within range of the remove_accent table
2241
19.8k
              if ((p[-2] != ' ') || (p[n] != ' ')) {
2242
                // not the only letter in the word
2243
8.80k
                p2 = p-1;
2244
8.80k
                p[-1] = ix;
2245
237k
                while ((p[0] = p[n]) != ' ')  p++;
2246
17.6k
                while (n-- > 0) *p++ = ' '; // replacement character must be no longer than original
2247
2248
8.80k
                if (tr->langopts.param[LOPT_DIERESES] && (lookupwchar(diereses_list, letter) > 0)) {
2249
                  // vowel with dieresis, replace and continue from this point
2250
0
                  p = p2;
2251
0
                  continue;
2252
0
                }
2253
2254
8.80k
                phonemes[0] = 0; // delete any phonemes which have been produced so far
2255
8.80k
                p = p_start;
2256
8.80k
                tr->word_vowel_count = 0;
2257
8.80k
                tr->word_stressed_count = 0;
2258
8.80k
                continue; // start again at the beginning of the word
2259
8.80k
              }
2260
19.8k
            }
2261
2262
1.14M
            if (((alphabet = AlphabetFromChar(letter)) != NULL)  && (alphabet->offset != tr->letter_bits_offset)) {
2263
106k
              if (tr->langopts.alt_alphabet == alphabet->offset) {
2264
340
                sprintf(phonemes, "%c%s", phonSWITCH, WordToString2(word_buf, tr->langopts.alt_alphabet_lang));
2265
340
                return 0;
2266
340
              }
2267
105k
              if (alphabet->flags & AL_WORDS) {
2268
                // switch to the nominated language for this alphabet
2269
59.0k
                sprintf(phonemes, "%c%s", phonSWITCH, WordToString2(word_buf, alphabet->language));
2270
59.0k
                return 0;
2271
59.0k
              }
2272
105k
            }
2273
1.14M
          }
2274
1.27M
        }
2275
2276
1.87M
        if (match1.points == 0) {
2277
1.17M
          if ((wc >= 0x300) && (wc <= 0x36f)) {
2278
            // combining accent inside a word, ignore
2279
1.16M
          } else if (IsAlpha(wc)) {
2280
278k
            if ((any_alpha > 1) || (p[wc_bytes-1] > ' ')) {
2281
              // an unrecognised character in a word, abort and then spell the word
2282
60.2k
              phonemes[0] = 0;
2283
60.2k
              if (dict_flags != NULL)
2284
15.1k
                dict_flags[0] |= FLAG_SPELLWORD;
2285
60.2k
              break;
2286
60.2k
            }
2287
887k
          } else {
2288
887k
            LookupLetter(tr, wc, -1, ph_buf, 0);
2289
887k
            if (ph_buf[0]) {
2290
84.5k
              match1.phonemes = ph_buf;
2291
84.5k
              match1.points = 1;
2292
84.5k
            }
2293
887k
          }
2294
1.11M
          p += (wc_bytes-1);
2295
1.11M
        } else
2296
700k
          tr->phonemes_repeat_count = 0;
2297
1.87M
      }
2298
2.19M
    }
2299
2300
1.98M
    if (match1.phonemes == NULL)
2301
43.3k
      match1.phonemes = "";
2302
2303
1.98M
    if (match1.points > 0) {
2304
954k
      if (word_flags & FLAG_UNPRON_TEST)
2305
7.09k
        return match1.end_type | 1;
2306
2307
947k
      if ((match1.phonemes[0] == phonSWITCH) && ((word_flags & FLAG_DONT_SWITCH_TRANSLATOR) == 0)) {
2308
        // an instruction to switch language, return immediately so we can re-translate
2309
29.1k
        strcpy(phonemes, match1.phonemes);
2310
29.1k
        return 0;
2311
29.1k
      }
2312
2313
918k
      if ((option_phonemes & espeakPHONEMES_TRACE) && ((word_flags & FLAG_NO_TRACE) == 0))
2314
0
        fprintf(f_trans, "\n");
2315
2316
918k
      match1.end_type &= ~SUFX_UNPRON;
2317
2318
918k
      if ((match1.end_type != 0) && (end_phonemes != NULL)) {
2319
        // a standard ending has been found, re-translate the word without it
2320
18.4k
        if ((match1.end_type & SUFX_P) && (word_flags & FLAG_NO_PREFIX)) {
2321
          // ignore the match on a prefix
2322
16.4k
        } else {
2323
16.4k
          if ((match1.end_type & SUFX_P) && ((match1.end_type & 0x7f) == 0)) {
2324
            // no prefix length specified
2325
16
            match1.end_type |= p - p_start;
2326
16
          }
2327
16.4k
          strcpy(end_phonemes, match1.phonemes);
2328
16.4k
          memcpy(p_start, word_copy, strlen(word_copy));
2329
16.4k
          return match1.end_type;
2330
16.4k
        }
2331
18.4k
      }
2332
902k
      if (match1.del_fwd != NULL)
2333
3.15k
        *match1.del_fwd = REPLACED_E;
2334
902k
      AppendPhonemes(tr, phonemes, ph_size, match1.phonemes);
2335
902k
    }
2336
1.98M
  }
2337
2338
813k
  memcpy(p_start, word_copy, strlen(word_copy));
2339
2340
813k
  return 0;
2341
1.00M
}
2342
2343
int TransposeAlphabet(Translator *tr, char *text)
2344
5.47M
{
2345
  // transpose cyrillic alphabet (for example) into ascii (single byte) character codes
2346
  // return: number of bytes, bit 6: 1=used compression
2347
2348
5.47M
  int c;
2349
5.47M
  int offset;
2350
5.47M
  int min;
2351
5.47M
  int max;
2352
5.47M
  const char *map;
2353
5.47M
  char *p = text;
2354
5.47M
  char *p2;
2355
5.47M
  bool all_alpha = true;
2356
5.47M
  int pairs_start;
2357
5.47M
  int bufix;
2358
5.47M
  char buf[N_WORD_BYTES+1];
2359
2360
5.47M
  offset = tr->transpose_min - 1;
2361
5.47M
  min = tr->transpose_min;
2362
5.47M
  max = tr->transpose_max;
2363
5.47M
  map = tr->transpose_map;
2364
2365
5.47M
  pairs_start = max - min + 2;
2366
2367
5.47M
  bufix = 0;
2368
6.75M
  do {
2369
6.75M
    p += utf8_in(&c, p);
2370
6.75M
    if (c != 0) {
2371
6.20M
      if ((c >= min) && (c <= max)) {
2372
1.47M
        if (map == NULL)
2373
15.8k
          buf[bufix++] = c - offset;
2374
1.46M
        else {
2375
          // get the code from the transpose map
2376
1.46M
          if (map[c - min] > 0)
2377
1.26M
            buf[bufix++] = map[c - min];
2378
195k
          else {
2379
195k
            all_alpha = false;
2380
195k
            break;
2381
195k
          }
2382
1.46M
        }
2383
4.72M
      } else {
2384
4.72M
        all_alpha = false;
2385
4.72M
        break;
2386
4.72M
      }
2387
6.20M
    }
2388
6.75M
  } while ((c != 0) && (bufix < N_WORD_BYTES));
2389
5.47M
  buf[bufix] = 0;
2390
2391
5.47M
  if (all_alpha) {
2392
    // compress to 6 bits per character
2393
549k
    int ix;
2394
549k
    int acc = 0;
2395
549k
    int bits = 0;
2396
2397
549k
    p = buf;
2398
549k
    p2 = buf;
2399
1.79M
    while ((c = *p++) != 0) {
2400
1.24M
      const short *pairs_list;
2401
1.24M
      if ((pairs_list = tr->frequent_pairs) != NULL) {
2402
10.5k
        int c2 = c + (*p << 8);
2403
148k
        for (ix = 0; c2 >= pairs_list[ix]; ix++) {
2404
140k
          if (c2 == pairs_list[ix]) {
2405
            // found an encoding for a 2-character pair
2406
2.54k
            c = ix + pairs_start; // 2-character codes start after the single letter codes
2407
2.54k
            p++;
2408
2.54k
            break;
2409
2.54k
          }
2410
140k
        }
2411
10.5k
      }
2412
1.24M
      acc = (acc << 6) + (c & 0x3f);
2413
1.24M
      bits += 6;
2414
2415
1.24M
      if (bits >= 8) {
2416
590k
        bits -= 8;
2417
590k
        *p2++ = (char)((acc >> bits) & 0xff);
2418
590k
      }
2419
1.24M
    }
2420
549k
    if (bits > 0)
2421
501k
      *p2++ = (char)((acc << (8-bits)) & 0xff);
2422
549k
    *p2 = 0;
2423
549k
    ix = p2 - buf;
2424
549k
    memcpy(text, buf, ix);
2425
549k
    return ix | 0x40; // bit 6 indicates compressed characters
2426
549k
  }
2427
4.92M
  return strlen(text);
2428
5.47M
}
2429
2430
/* Find an entry in the word_dict file for a specified word.
2431
   Returns NULL if no match, else returns 'word_end'
2432
2433
    word   zero terminated word to match
2434
    word2  pointer to next word(s) in the input text (terminated by space)
2435
2436
    flags:  returns dictionary flags which are associated with a matched word
2437
2438
    end_flags:  indicates whether this is a retranslation after removing a suffix
2439
 */
2440
static const char *LookupDict2(Translator *tr, const char *word, const char *word2,
2441
                               char *phonetic, unsigned int *flags, int end_flags, WORD_TAB *wtab, int wtab_remaining)
2442
5.47M
{
2443
5.47M
  char *p;
2444
5.47M
  char *next;
2445
5.47M
  int hash;
2446
5.47M
  int phoneme_len;
2447
5.47M
  int wlen;
2448
5.47M
  unsigned char flag;
2449
5.47M
  unsigned int dictionary_flags;
2450
5.47M
  unsigned int dictionary_flags2;
2451
5.47M
  bool condition_failed = false;
2452
5.47M
  int n_chars;
2453
5.47M
  int no_phonemes;
2454
5.47M
  int skipwords;
2455
5.47M
  int ix;
2456
5.47M
  int c;
2457
5.47M
  const char *word_end;
2458
5.47M
  const char *word1;
2459
5.47M
  int wflags = 0;
2460
5.47M
  int lookup_symbol;
2461
5.47M
  char word_buf[N_WORD_BYTES+1];
2462
5.47M
  char dict_flags_buf[80];
2463
2464
5.47M
  if (wtab != NULL)
2465
1.20M
    wflags = wtab->flags;
2466
2467
5.47M
  lookup_symbol = flags[1] & FLAG_LOOKUP_SYMBOL;
2468
5.47M
  word1 = word;
2469
5.47M
  if (tr->transpose_min > 0) {
2470
5.47M
    strncpy0(word_buf, word, N_WORD_BYTES);
2471
5.47M
    wlen = TransposeAlphabet(tr, word_buf); // bit 6 indicates compressed characters
2472
5.47M
    word = word_buf;
2473
5.47M
  } else
2474
0
    wlen = strlen(word);
2475
2476
5.47M
  hash = HashDictionary(word);
2477
5.47M
  p = tr->dict_hashtab[hash];
2478
2479
5.47M
  if (p == NULL) {
2480
0
    if (flags != NULL)
2481
0
      *flags = 0;
2482
0
    return 0;
2483
0
  }
2484
2485
  // Find the first entry in the list for this hash value which matches.
2486
  // This corresponds to the last matching entry in the *_list file.
2487
2488
115M
  while (*p != 0) {
2489
110M
    next = p + (p[0] & 0xff);
2490
2491
110M
    if (((p[1] & 0x7f) != wlen) || (memcmp(word, &p[2], wlen & 0x3f) != 0)) {
2492
      // bit 6 of wlen indicates whether the word has been compressed; so we need to match on this also.
2493
108M
      p = next;
2494
108M
      continue;
2495
108M
    }
2496
2497
    // found matching entry. Decode the phonetic string
2498
2.09M
    word_end = word2;
2499
2500
2.09M
    dictionary_flags = 0;
2501
2.09M
    dictionary_flags2 = 0;
2502
2.09M
    no_phonemes = p[1] & 0x80;
2503
2504
2.09M
    p += ((p[1] & 0x3f) + 2);
2505
2506
2.09M
    if (no_phonemes) {
2507
27.8k
      phonetic[0] = 0;
2508
27.8k
      phoneme_len = 0;
2509
2.06M
    } else {
2510
2.06M
      phoneme_len = strlen(p);
2511
2.06M
      assert(phoneme_len < N_PHONEME_BYTES);
2512
2.06M
      strcpy(phonetic, p);
2513
2.06M
      p += (phoneme_len + 1);
2514
2.06M
    }
2515
2516
3.34M
    while (p < next) {
2517
      // examine the flags which follow the phoneme string
2518
2519
1.83M
      flag = *p++;
2520
1.83M
      if (flag >= 100) {
2521
        // conditional rule
2522
91.7k
        if (flag >= 132) {
2523
          // fail if this condition is set
2524
3.68k
          if ((tr->dict_condition & (1 << (flag-132))) != 0)
2525
0
            condition_failed = true;
2526
88.1k
        } else {
2527
          // allow only if this condition is set
2528
88.1k
          if ((tr->dict_condition & (1 << (flag-100))) == 0)
2529
75.9k
            condition_failed = true;
2530
88.1k
        }
2531
1.74M
      } else if (flag > 80) {
2532
        // flags 81 to 90  match more than one word
2533
        // This comes after the other flags
2534
583k
        n_chars = next - p;
2535
583k
        skipwords = flag - 80;
2536
2537
        // don't use the contraction if any of the words are emphasized
2538
        //  or has an embedded command, such as MARK
2539
583k
        if ((wtab != NULL) && (wtab_remaining > skipwords)) {
2540
1.42M
          for (ix = 0; ix <= skipwords && wtab[ix].length; ix++) {
2541
902k
            if (wtab[ix].flags & FLAG_EMPHASIZED2)
2542
24.6k
              condition_failed = true;
2543
2544
902k
          }
2545
521k
        }
2546
2547
583k
        if (strncmp(word2, p, n_chars) != 0)
2548
582k
          condition_failed = true;
2549
2550
583k
        if (condition_failed) {
2551
582k
          p = next;
2552
582k
          break;
2553
582k
        }
2554
2555
906
        dictionary_flags |= FLAG_SKIPWORDS;
2556
906
        dictionary_skipwords = skipwords;
2557
906
        p = next;
2558
906
        word_end = word2 + n_chars;
2559
1.15M
      } else if (flag > 64) {
2560
        // stressed syllable information, put in bits 0-3
2561
53.5k
        dictionary_flags = (dictionary_flags & ~0xf) | (flag & 0xf);
2562
53.5k
        if ((flag & 0xc) == 0xc)
2563
34.5k
          dictionary_flags |= FLAG_STRESS_END;
2564
1.10M
      } else if (flag >= 32)
2565
400k
        dictionary_flags2 |= (1L << (flag-32));
2566
703k
      else
2567
703k
        dictionary_flags |= (1L << flag);
2568
1.83M
    }
2569
2570
2.09M
    if (condition_failed) {
2571
652k
      condition_failed = false;
2572
652k
      continue;
2573
652k
    }
2574
2575
1.43M
    if ((end_flags & FLAG_SUFX) == 0) {
2576
      // no suffix has been removed
2577
1.43M
      if (dictionary_flags2 & FLAG_STEM)
2578
10
        continue; // this word must have a suffix
2579
1.43M
    }
2580
2581
1.43M
    if ((end_flags & SUFX_P) && (dictionary_flags2 & (FLAG_ONLY | FLAG_ONLY_S)))
2582
65
      continue; // $only or $onlys, don't match if a prefix has been removed
2583
2584
1.43M
    if (end_flags & FLAG_SUFX) {
2585
      // a suffix was removed from the word
2586
3.06k
      if (dictionary_flags2 & FLAG_ONLY)
2587
986
        continue; // no match if any suffix
2588
2589
2.07k
      if ((dictionary_flags2 & FLAG_ONLY_S) && ((end_flags & FLAG_SUFX_S) == 0)) {
2590
        // only a 's' suffix allowed, but the suffix wasn't 's'
2591
104
        continue;
2592
104
      }
2593
2.07k
    }
2594
2595
1.43M
    if (dictionary_flags2 & FLAG_CAPITAL) {
2596
232
      if (!(wflags & FLAG_FIRST_UPPER))
2597
31
        continue;
2598
232
    }
2599
1.43M
    if (dictionary_flags2 & FLAG_ALLCAPS) {
2600
22.1k
      if (!(wflags & FLAG_ALL_UPPER))
2601
17.5k
        continue;
2602
22.1k
    }
2603
1.41M
    if (dictionary_flags & FLAG_NEEDS_DOT) {
2604
15
      if (!(wflags & FLAG_HAS_DOT))
2605
15
        continue;
2606
15
    }
2607
2608
1.41M
    if ((dictionary_flags2 & FLAG_ATEND) && (word_end < translator->clause_end) && (lookup_symbol == 0)) {
2609
      // only use this pronunciation if it's the last word of the clause, or called from Lookup()
2610
10.8k
      continue;
2611
10.8k
    }
2612
2613
1.40M
    if ((dictionary_flags2 & FLAG_ATSTART) && !(wflags & FLAG_FIRST_WORD)) {
2614
      // only use this pronunciation if it's the first word of a clause
2615
52
      continue;
2616
52
    }
2617
2618
1.40M
    if ((dictionary_flags2 & FLAG_SENTENCE) && !(translator->clause_terminator & CLAUSE_TYPE_SENTENCE)) {
2619
      // only if this clause is a sentence , i.e. terminator is {. ? !} not {, : :}
2620
103
      continue;
2621
103
    }
2622
2623
1.40M
    if (dictionary_flags2 & FLAG_VERB) {
2624
      // this is a verb-form pronunciation
2625
2626
1.17k
      if (tr->expect_verb || (tr->expect_verb_s && (end_flags & FLAG_SUFX_S))) {
2627
        // OK, we are expecting a verb
2628
266
        if ((tr->translator_name == L('e', 'n')) && (tr->prev_dict_flags[0] & FLAG_ALT7_TRANS) && (end_flags & FLAG_SUFX_S)) {
2629
          // lang=en, don't use verb form after 'to' if the word has 's' suffix
2630
10
          continue;
2631
10
        }
2632
912
      } else {
2633
        // don't use the 'verb' pronunciation unless we are expecting a verb
2634
912
        continue;
2635
912
      }
2636
1.17k
    }
2637
1.40M
    if (dictionary_flags2 & FLAG_PAST) {
2638
167
      if (!tr->expect_past) {
2639
        // don't use the 'past' pronunciation unless we are expecting past tense
2640
160
        continue;
2641
160
      }
2642
167
    }
2643
1.40M
    if (dictionary_flags2 & FLAG_NOUN) {
2644
25
      if ((!tr->expect_noun) || (end_flags & SUFX_V)) {
2645
        // don't use the 'noun' pronunciation unless we are expecting a noun
2646
25
        continue;
2647
25
      }
2648
25
    }
2649
1.40M
    if (dictionary_flags2 & FLAG_NATIVE) {
2650
247
      if (tr != translator)
2651
246
        continue; // don't use if we've switched translators
2652
247
    }
2653
1.40M
    if (dictionary_flags & FLAG_ALT2_TRANS) {
2654
      // language specific
2655
958
      if ((tr->translator_name == L('h', 'u')) && !(tr->prev_dict_flags[0] & FLAG_ALT_TRANS))
2656
795
        continue;
2657
958
    }
2658
2659
1.40M
    if (flags != NULL) {
2660
1.40M
      flags[0] = dictionary_flags | FLAG_FOUND_ATTRIBUTES;
2661
1.40M
      flags[1] = dictionary_flags2;
2662
1.40M
    }
2663
2664
1.40M
    if (phoneme_len == 0) {
2665
16.1k
      if (option_phonemes & espeakPHONEMES_TRACE) {
2666
0
        print_dictionary_flags(flags, dict_flags_buf, sizeof(dict_flags_buf));
2667
0
        fprintf(f_trans, "Flags:  %s  %s\n", word1, dict_flags_buf);
2668
0
      }
2669
16.1k
      return 0; // no phoneme translation found here, only flags. So use rules
2670
16.1k
    }
2671
2672
1.38M
    if (flags != NULL)
2673
1.38M
      flags[0] |= FLAG_FOUND; // this flag indicates word was found in dictionary
2674
2675
1.38M
    if (option_phonemes & espeakPHONEMES_TRACE) {
2676
0
      char ph_decoded[N_WORD_PHONEMES];
2677
0
      bool textmode;
2678
2679
0
      DecodePhonemes(phonetic, ph_decoded);
2680
2681
0
      if ((dictionary_flags & FLAG_TEXTMODE) == 0)
2682
0
        textmode = false;
2683
0
      else
2684
0
        textmode = true;
2685
2686
0
      if (textmode == translator->langopts.textmode) {
2687
        // only show this line if the word translates to phonemes, not replacement text
2688
0
        if ((dictionary_flags & FLAG_SKIPWORDS) && (wtab != NULL)) {
2689
          // matched more than one word
2690
          // (check for wtab prevents showing RULE_SPELLING byte when speaking individual letters)
2691
0
          memcpy(word_buf, word2, word_end-word2);
2692
0
          word_buf[word_end-word2-1] = 0;
2693
0
          fprintf(f_trans, "Found: '%s %s\n", word1, word_buf);
2694
0
        } else
2695
0
          fprintf(f_trans, "Found: '%s", word1);
2696
0
        print_dictionary_flags(flags, dict_flags_buf, sizeof(dict_flags_buf));
2697
0
        fprintf(f_trans, "' [%s]  %s\n", ph_decoded, dict_flags_buf);
2698
0
      }
2699
0
    }
2700
2701
1.38M
    ix = utf8_in(&c, word);
2702
1.38M
    if (flags != NULL && (word[ix] == 0) && !IsAlpha(c))
2703
474k
      flags[0] |= FLAG_MAX3;
2704
1.38M
    return word_end;
2705
2706
1.40M
  }
2707
4.06M
  return 0;
2708
5.47M
}
2709
2710
2711
    static int utf8_nbytes(const char *buf)
2712
5.53M
{
2713
  // Returns the number of bytes for the first UTF-8 character in buf
2714
2715
5.53M
  unsigned char c = (unsigned char)buf[0];
2716
5.53M
  if (c < 0x80)
2717
4.84M
    return 1;
2718
691k
  if (c < 0xe0)
2719
168k
    return 2;
2720
522k
  if (c < 0xf0)
2721
411k
    return 3;
2722
111k
  return 4;
2723
522k
}
2724
2725
/* Lookup a specified word in the word dictionary.
2726
   Returns phonetic data in 'phonetic' and bits in 'flags'
2727
2728
   end_flags:  indicates if a suffix has been removed
2729
 */
2730
int LookupDictList(Translator *tr, char **wordptr, char *ph_out, unsigned int *flags, int end_flags, WORD_TAB *wtab, int wtab_remaining)
2731
5.45M
{
2732
5.45M
  int length;
2733
5.45M
  const char *found;
2734
5.45M
  const char *word1;
2735
5.45M
  const char *word2;
2736
5.45M
  unsigned char c;
2737
5.45M
  int nbytes;
2738
5.45M
  int len;
2739
5.45M
  char word[N_WORD_BYTES];
2740
5.45M
  static char word_replacement[N_WORD_BYTES];
2741
2742
5.45M
  MAKE_MEM_UNDEFINED(&word_replacement, sizeof(word_replacement));
2743
2744
5.45M
  length = 0;
2745
5.45M
  word2 = word1 = *wordptr;
2746
2747
5.53M
  while ((word2[nbytes = utf8_nbytes(word2)] == ' ') && (word2[nbytes+1] == '.')) {
2748
    // look for an abbreviation of the form a.b.c
2749
    // try removing the spaces between the dots and looking for a match
2750
81.7k
    if ((nbytes <= 0) || ((size_t)nbytes + 1 > sizeof(word) - (size_t)length)) {
2751
      /* Too long abbreviation, leave as it is */
2752
374
      length = 0;
2753
374
      break;
2754
374
    }
2755
81.4k
    memcpy(&word[length], word2, nbytes);
2756
81.4k
    length += nbytes;
2757
81.4k
    word[length++] = '.';
2758
81.4k
    word2 += nbytes+3;
2759
81.4k
  }
2760
5.45M
  if (length > 0) {
2761
    // found an abbreviation containing dots
2762
13.3k
    nbytes = 0;
2763
97.4k
    while (((c = word2[nbytes]) != 0) && (c != ' '))
2764
84.1k
      nbytes++;
2765
13.3k
    if (length + nbytes + 1 <= sizeof(word)) {
2766
13.1k
      memcpy(&word[length], word2, nbytes);
2767
13.1k
      word[length+nbytes] = 0;
2768
13.1k
      found =  LookupDict2(tr, word, word2, ph_out, flags, end_flags, wtab, wtab_remaining);
2769
13.1k
      if (found) {
2770
        // set the skip words flag
2771
270
        flags[0] |= FLAG_SKIPWORDS;
2772
270
        dictionary_skipwords = length;
2773
270
        return 1;
2774
270
      }
2775
13.1k
    }
2776
13.3k
  }
2777
2778
18.5M
  for (length = 0; length < (N_WORD_BYTES-1); length++) {
2779
18.5M
    if (((c = *word1++) == 0) || (c == ' '))
2780
5.43M
      break;
2781
2782
13.1M
    if ((c == '.') && (length > 0) && (IsDigit09(word[length-1])))
2783
23.6k
      break; // needed for lang=hu, eg. "december 2.-ig"
2784
2785
13.1M
    word[length] = c;
2786
13.1M
  }
2787
5.45M
  word[length] = 0;
2788
2789
5.45M
  found = LookupDict2(tr, word, word1, ph_out, flags, end_flags, wtab, wtab_remaining);
2790
2791
5.45M
  if (flags[0] & FLAG_MAX3) {
2792
479k
    if (strcmp(ph_out, tr->phonemes_repeat) == 0) {
2793
229k
      tr->phonemes_repeat_count++;
2794
229k
      if (tr->phonemes_repeat_count > 3)
2795
90.6k
        ph_out[0] = 0;
2796
250k
    } else {
2797
250k
      strncpy0(tr->phonemes_repeat, ph_out, sizeof(tr->phonemes_repeat));
2798
250k
      tr->phonemes_repeat_count = 1;
2799
250k
    }
2800
479k
  } else
2801
4.97M
    tr->phonemes_repeat_count = 0;
2802
2803
5.45M
  if ((found == 0) && (flags[1] & FLAG_ACCENT)) {
2804
120
    int letter;
2805
120
    word2 = word;
2806
120
    if (*word2 == '_') word2++;
2807
120
    len = utf8_in(&letter, word2);
2808
120
    LookupAccentedLetter(tr, letter, ph_out);
2809
120
    found = word2 + len;
2810
120
  }
2811
2812
5.45M
  if (found == 0 && length >= 2) {
2813
2.58M
    ph_out[0] = 0;
2814
2815
    // try modifications to find a recognised word
2816
2817
2.58M
    if ((end_flags & FLAG_SUFX_E_ADDED) && (word[length-1] == 'e')) {
2818
      // try removing an 'e' which has been added by RemoveEnding
2819
1.77k
      word[length-1] = 0;
2820
1.77k
      found = LookupDict2(tr, word, word1, ph_out, flags, end_flags, wtab, wtab_remaining);
2821
2.58M
    } else if ((end_flags & SUFX_D) && (word[length-1] == word[length-2])) {
2822
      // try removing a double letter
2823
434
      word[length-1] = 0;
2824
434
      found = LookupDict2(tr, word, word1, ph_out, flags, end_flags, wtab, wtab_remaining);
2825
434
    }
2826
2.58M
  }
2827
2828
5.45M
  if (found) {
2829
    // if textmode is the default, then words which have phonemes are marked.
2830
1.38M
    if (tr->langopts.textmode)
2831
4.24k
      *flags ^= FLAG_TEXTMODE;
2832
2833
1.38M
    if (*flags & FLAG_TEXTMODE) {
2834
      // the word translates to replacement text, not to phonemes
2835
2836
35.2k
      if (end_flags & FLAG_ALLOW_TEXTMODE) {
2837
        // only use replacement text if this is the original word, not if a prefix or suffix has been removed
2838
34.5k
        word_replacement[0] = 0;
2839
34.5k
        word_replacement[1] = ' ';
2840
34.5k
        sprintf(&word_replacement[2], "%s ", ph_out); // replacement word, preceded by zerochar and space
2841
2842
34.5k
        word1 = *wordptr;
2843
34.5k
        *wordptr = &word_replacement[2];
2844
2845
34.5k
        if (option_phonemes & espeakPHONEMES_TRACE) {
2846
0
          len = found - word1;
2847
0
          memcpy(word, word1, len); // include multiple matching words
2848
0
          word[len] = 0;
2849
0
          fprintf(f_trans, "Replace: %s  %s\n", word, *wordptr);
2850
0
        }
2851
34.5k
      }
2852
2853
35.2k
      ph_out[0] = 0;
2854
35.2k
      return 0;
2855
35.2k
    }
2856
2857
1.35M
    return 1;
2858
1.38M
  }
2859
2860
4.06M
  ph_out[0] = 0;
2861
4.06M
  return 0;
2862
5.45M
}
2863
2864
extern char word_phonemes[N_WORD_PHONEMES]; // a word translated into phoneme codes
2865
2866
int Lookup(Translator *tr, const char *word, char *ph_out)
2867
4.16M
{
2868
  // Look up in *_list, returns dictionary flags[0] and phonemes
2869
2870
4.16M
  int flags0;
2871
4.16M
  unsigned int flags[2];
2872
4.16M
  char *word1 = (char *)word;
2873
2874
4.16M
  flags[0] = 0;
2875
4.16M
  flags[1] = FLAG_LOOKUP_SYMBOL;
2876
4.16M
  if ((flags0 = LookupDictList(tr, &word1, ph_out, flags, FLAG_ALLOW_TEXTMODE, NULL, 0)) != 0)
2877
1.15M
    flags0 = flags[0];
2878
2879
4.16M
  if (flags[0] & FLAG_TEXTMODE) {
2880
18.0k
    int say_as = option_sayas;
2881
18.0k
    option_sayas = 0; // don't speak replacement word as letter names
2882
    // NOTE: TranslateRoman checks text[-2] and IsLetterGroup looks
2883
    // for a heading \0, so pad the start of text to prevent
2884
    // it reading data on the stack.
2885
18.0k
    char text[80];
2886
2887
18.0k
    text[0] = 0;
2888
18.0k
    text[1] = ' ';
2889
18.0k
    text[2] = ' ';
2890
18.0k
    strncpy0(text+3, word1, sizeof(text)-3);
2891
18.0k
    flags0 = TranslateWord(tr, text+3, NULL, NULL);
2892
18.0k
    strcpy(ph_out, word_phonemes);
2893
18.0k
    option_sayas = say_as;
2894
18.0k
  }
2895
4.16M
  return flags0;
2896
4.16M
}
2897
2898
static int LookupFlags(Translator *tr, const char *word, unsigned int flags_out[2])
2899
674
{
2900
674
  char buf[100];
2901
674
  static unsigned int flags[2];
2902
674
  char *word1 = (char *)word;
2903
2904
674
  flags[0] = flags[1] = 0;
2905
674
  LookupDictList(tr, &word1, buf, flags, 0, NULL, 0);
2906
674
  flags_out[0] = flags[0];
2907
674
  flags_out[1] = flags[1];
2908
674
  return flags[0];
2909
674
}
2910
2911
int RemoveEnding(Translator *tr, char *word, int end_type, char *word_copy)
2912
13.7k
{
2913
  /* Removes a standard suffix from a word, once it has been indicated by the dictionary rules.
2914
     end_type: bits 0-6  number of letters
2915
               bits 8-14  suffix flags
2916
2917
      word_copy: make a copy of the original word
2918
      This routine is language specific.  In English it deals with reversing y->i and e-dropping
2919
      that were done when the suffix was added to the original word.
2920
   */
2921
2922
13.7k
  int i;
2923
13.7k
  char *word_end;
2924
13.7k
  int len_ending;
2925
13.7k
  int end_flags;
2926
13.7k
  char ending[50] = {0};
2927
2928
  // these lists are language specific, but are only relevant if the 'e' suffix flag is used
2929
13.7k
  static const char * const add_e_exceptions[] = {
2930
13.7k
    "ion", NULL
2931
13.7k
  };
2932
2933
13.7k
  static const char * const add_e_additions[] = {
2934
13.7k
    "c", "rs", "ir", "ur", "ath", "ns", "u",
2935
13.7k
    "spong", // sponge
2936
13.7k
    "rang", // strange
2937
13.7k
    "larg", // large
2938
13.7k
    NULL
2939
13.7k
  };
2940
2941
118k
  for (word_end = word; *word_end != ' '; word_end++) {
2942
    // replace discarded 'e's
2943
104k
    if (*word_end == REPLACED_E)
2944
109
      *word_end = 'e';
2945
104k
  }
2946
13.7k
  i = word_end - word;
2947
13.7k
  if (i >= N_WORD_BYTES) i = N_WORD_BYTES-1;
2948
2949
13.7k
  if (word_copy != NULL) {
2950
13.4k
    memcpy(word_copy, word, i);
2951
13.4k
    word_copy[i] = 0;
2952
13.4k
  }
2953
2954
  // look for multibyte characters to increase the number of bytes to remove
2955
37.1k
  for (len_ending = i = (end_type & 0x3f); i > 0; i--) { // num.of characters of the suffix
2956
23.4k
    word_end--;
2957
23.8k
    while (word_end >= word && (*word_end & 0xc0) == 0x80) {
2958
410
      word_end--; // for multibyte characters
2959
410
      len_ending++;
2960
410
    }
2961
23.4k
  }
2962
2963
  // remove bytes from the end of the word and replace them by spaces
2964
37.6k
  for (i = 0; (i < len_ending) && (i < (int)sizeof(ending)-1); i++) {
2965
23.8k
    ending[i] = word_end[i];
2966
23.8k
    word_end[i] = ' ';
2967
23.8k
  }
2968
13.7k
  ending[i] = 0;
2969
13.7k
  word_end--; // now pointing at last character of stem
2970
2971
13.7k
  end_flags = (end_type & 0xfff0) | FLAG_SUFX;
2972
2973
  /* add an 'e' to the stem if appropriate,
2974
      if  stem ends in vowel+consonant
2975
      or  stem ends in 'c'  (add 'e' to soften it) */
2976
2977
13.7k
  if (end_type & SUFX_I) {
2978
1.59k
    if (word_end[0] == 'i')
2979
94
      word_end[0] = 'y';
2980
1.59k
  }
2981
2982
13.7k
  if (end_type & SUFX_E) {
2983
2.85k
    if (tr->translator_name == L('n', 'l')) {
2984
123
      if (((word_end[0] & 0x80) == 0) && ((word_end[-1] & 0x80) == 0) && IsVowel(tr, word_end[-1]) && IsLetter(tr, word_end[0], LETTERGP_C) && !IsVowel(tr, word_end[-2])) {
2985
        // double the vowel before the (ascii) final consonant
2986
113
        word_end[1] = word_end[0];
2987
113
        word_end[0] = word_end[-1];
2988
113
        word_end[2] = ' ';
2989
113
      }
2990
2.72k
    } else if (tr->translator_name == L('e', 'n')) {
2991
      // add 'e' to end of stem
2992
2.72k
      if (IsLetter(tr, word_end[-1], LETTERGP_VOWEL2) && IsLetter(tr, word_end[0], 1)) {
2993
        // vowel(incl.'y') + hard.consonant
2994
2995
1.92k
        const char *p;
2996
3.85k
        for (i = 0; (p = add_e_exceptions[i]) != NULL; i++) {
2997
1.92k
          int len = strlen(p);
2998
1.92k
          if (word_end + 1-len >= word && memcmp(p, &word_end[1-len], len) == 0)
2999
0
            break;
3000
1.92k
        }
3001
1.92k
        if (p == NULL)
3002
1.92k
          end_flags |= FLAG_SUFX_E_ADDED; // no exception found
3003
1.92k
      } else {
3004
800
        const char *p;
3005
8.56k
        for (i = 0; (p = add_e_additions[i]) != NULL; i++) {
3006
7.78k
          int len = strlen(p);
3007
7.78k
          if (word_end + 1-len >= word && memcmp(p, &word_end[1-len], len) == 0) {
3008
25
            end_flags |= FLAG_SUFX_E_ADDED;
3009
25
            break;
3010
25
          }
3011
7.78k
        }
3012
800
      }
3013
2.72k
    } else if (tr->langopts.suffix_add_e != 0)
3014
0
      end_flags |= FLAG_SUFX_E_ADDED;
3015
3016
2.85k
    if (end_flags & FLAG_SUFX_E_ADDED) {
3017
1.95k
      utf8_out(tr->langopts.suffix_add_e, &word_end[1]);
3018
3019
1.95k
      if (option_phonemes & espeakPHONEMES_TRACE)
3020
0
        fprintf(f_trans, "add e\n");
3021
1.95k
    }
3022
2.85k
  }
3023
3024
13.7k
  if ((end_type & SUFX_V) && (tr->expect_verb == 0))
3025
3.48k
    tr->expect_verb = 1; // this suffix indicates the verb pronunciation
3026
3027
3028
13.7k
  if ((strcmp(ending, "s") == 0) || (strcmp(ending, "es") == 0))
3029
4.16k
    end_flags |= FLAG_SUFX_S;
3030
3031
13.7k
  if (ending[0] == '\'')
3032
1.27k
    end_flags &= ~FLAG_SUFX; // don't consider 's as an added suffix
3033
3034
13.7k
  return end_flags;
3035
13.7k
}
3036
3037
674
static void DollarRule(char *word[], char *word_start, int consumed, int group_length, char word_buf[N_WORD_BYTES], Translator *tr, int command, int *failed, int *add_points) {
3038
  // $list or $p_alt
3039
  // make a copy of the word up to the post-match characters
3040
674
  int ix = *word - word_start + consumed + group_length + 1;
3041
3042
674
  if (ix+2 > N_WORD_BYTES) {
3043
0
    *failed = 1;
3044
0
    return;
3045
0
  }
3046
3047
674
  memcpy(word_buf, word_start-1, ix);
3048
674
  word_buf[ix] = ' ';
3049
674
  word_buf[ix+1] = 0;
3050
674
  unsigned int flags[2];
3051
674
  LookupFlags(tr, &word_buf[1], flags);
3052
3053
674
  if ((command == DOLLAR_LIST) && (flags[0] & FLAG_FOUND) && !(flags[1] & FLAG_ONLY))
3054
0
    *add_points = 23;
3055
674
  else if (flags[0] & (1 << (BITNUM_FLAG_ALT + (command & 0xf))))
3056
271
    *add_points = 23;
3057
403
  else
3058
403
    *failed = 1;
3059
674
}