Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/babel/messages/catalog.py: 35%

Shortcuts on this page

r m x   toggle line displays

j k   next/prev highlighted chunk

0   (zero) top of page

1   (one) first highlighted chunk

448 statements  

1""" 

2babel.messages.catalog 

3~~~~~~~~~~~~~~~~~~~~~~ 

4 

5Data structures for message catalogs. 

6 

7:copyright: (c) 2013-2026 by the Babel Team. 

8:license: BSD, see LICENSE for more details. 

9""" 

10 

11from __future__ import annotations 

12 

13import datetime 

14import re 

15from collections import defaultdict 

16from collections.abc import Iterable, Iterator 

17from copy import copy 

18from difflib import SequenceMatcher 

19from email import message_from_string 

20from heapq import nlargest 

21from string import Formatter 

22from typing import TYPE_CHECKING, TypedDict 

23 

24from babel import __version__ as VERSION 

25from babel.core import Locale, UnknownLocaleError 

26from babel.dates import format_datetime 

27from babel.messages.plurals import get_plural 

28from babel.util import LOCALTZ, _cmp 

29 

30if TYPE_CHECKING: 

31 from typing_extensions import TypeAlias 

32 

33 _MessageID: TypeAlias = str | tuple[str, ...] | list[str] 

34 

35__all__ = [ 

36 'DEFAULT_HEADER', 

37 'PYTHON_FORMAT', 

38 'Catalog', 

39 'Message', 

40 'TranslationError', 

41] 

42 

43 

44def get_close_matches(word, possibilities, n=3, cutoff=0.6): 

45 """A modified version of ``difflib.get_close_matches``. 

46 

47 It just passes ``autojunk=False`` to the ``SequenceMatcher``, to work 

48 around https://github.com/python/cpython/issues/90825. 

49 """ 

50 if not n > 0: # pragma: no cover 

51 raise ValueError(f"n must be > 0: {n!r}") 

52 if not 0.0 <= cutoff <= 1.0: # pragma: no cover 

53 raise ValueError(f"cutoff must be in [0.0, 1.0]: {cutoff!r}") 

54 result = [] 

55 s = SequenceMatcher(autojunk=False) # only line changed from difflib.py 

56 s.set_seq2(word) 

57 for x in possibilities: 

58 s.set_seq1(x) 

59 if ( 

60 s.real_quick_ratio() >= cutoff 

61 and s.quick_ratio() >= cutoff 

62 and s.ratio() >= cutoff 

63 ): 

64 result.append((s.ratio(), x)) 

65 

66 # Move the best scorers to head of list 

67 result = nlargest(n, result) 

68 # Strip scores for the best n matches 

69 return [x for score, x in result] 

70 

71 

72PYTHON_FORMAT = re.compile( 

73 r''' 

74 \% 

75 (?:\(([\w]*)\))? 

76 ( 

77 [-#0\ +]?(?:\*|[\d]+)? 

78 (?:\.(?:\*|[\d]+))? 

79 [hlL]? 

80 ) 

81 ([diouxXeEfFgGcrs%]) 

82''', 

83 re.VERBOSE, 

84) 

85 

86 

87def _has_python_brace_format(string: str) -> bool: 

88 if "{" not in string: 

89 return False 

90 fmt = Formatter() 

91 try: 

92 # `fmt.parse` returns 3-or-4-tuples of the form 

93 # `(literal_text, field_name, format_spec, conversion)`; 

94 # if `field_name` is set, this smells like brace format 

95 field_name_seen = False 

96 for t in fmt.parse(string): 

97 if t[1] is not None: 

98 field_name_seen = True 

99 # We cannot break here, as we need to consume the whole string 

100 # to ensure that it is a valid format string. 

101 except ValueError: 

102 return False 

103 return field_name_seen 

104 

105 

106def _parse_datetime_header(value: str) -> datetime.datetime: 

107 match = re.match(r'^(?P<datetime>.*?)(?P<tzoffset>[+-]\d{4})?$', value) 

108 

109 dt = datetime.datetime.strptime(match.group('datetime'), '%Y-%m-%d %H:%M') 

110 

111 # Separate the offset into a sign component, hours, and # minutes 

112 tzoffset = match.group('tzoffset') 

113 if tzoffset is not None: 

114 plus_minus_s, rest = tzoffset[0], tzoffset[1:] 

115 hours_offset_s, mins_offset_s = rest[:2], rest[2:] 

116 

117 # Make them all integers 

118 plus_minus = int(f"{plus_minus_s}1") 

119 hours_offset = int(hours_offset_s) 

120 mins_offset = int(mins_offset_s) 

121 

122 # Calculate net offset 

123 net_mins_offset = hours_offset * 60 

124 net_mins_offset += mins_offset 

125 net_mins_offset *= plus_minus 

126 

127 # Create an offset object 

128 tzoffset = datetime.timezone( 

129 offset=datetime.timedelta(minutes=net_mins_offset), 

130 name=f'Etc/GMT{net_mins_offset:+d}', 

131 ) 

132 

133 # Store the offset in a datetime object 

134 dt = dt.replace(tzinfo=tzoffset) 

135 

136 return dt 

137 

138 

139class Message: 

140 """Representation of a single message in a catalog.""" 

141 

142 def __init__( 

143 self, 

144 id: _MessageID, 

145 string: _MessageID | None = '', 

146 locations: Iterable[tuple[str, int]] = (), 

147 flags: Iterable[str] = (), 

148 auto_comments: Iterable[str] = (), 

149 user_comments: Iterable[str] = (), 

150 previous_id: _MessageID = (), 

151 lineno: int | None = None, 

152 context: str | None = None, 

153 ) -> None: 

154 """Create the message object. 

155 

156 :param id: the message ID, or a ``(singular, plural)`` tuple for 

157 pluralizable messages 

158 :param string: the translated message string, or a 

159 ``(singular, plural)`` tuple for pluralizable messages 

160 :param locations: a sequence of ``(filename, lineno)`` tuples 

161 :param flags: a set or sequence of flags 

162 :param auto_comments: a sequence of automatic comments for the message 

163 :param user_comments: a sequence of user comments for the message 

164 :param previous_id: the previous message ID, or a ``(singular, plural)`` 

165 tuple for pluralizable messages 

166 :param lineno: the line number on which the msgid line was found in the 

167 PO file, if any 

168 :param context: the message context 

169 """ 

170 self.id = id 

171 if not string and self.pluralizable: 

172 string = ('', '') 

173 self.string = string 

174 self.locations = list(dict.fromkeys(locations)) if locations else [] 

175 self.flags = set(flags) 

176 if id and self.python_format: 

177 self.flags.add('python-format') 

178 else: 

179 self.flags.discard('python-format') 

180 if id and self.python_brace_format: 

181 self.flags.add('python-brace-format') 

182 else: 

183 self.flags.discard('python-brace-format') 

184 self.auto_comments = list(dict.fromkeys(auto_comments)) if auto_comments else [] 

185 self.user_comments = list(dict.fromkeys(user_comments)) if user_comments else [] 

186 if previous_id: 

187 if isinstance(previous_id, str): 

188 self.previous_id = [previous_id] 

189 else: 

190 self.previous_id = list(previous_id) 

191 else: 

192 self.previous_id = [] 

193 self.lineno = lineno 

194 self.context = context 

195 

196 def __repr__(self) -> str: 

197 return f"<{type(self).__name__} {self.id!r} (flags: {list(self.flags)!r})>" 

198 

199 def __cmp__(self, other: object) -> int: 

200 """Compare Messages, taking into account plural ids""" 

201 

202 def values_to_compare(obj): 

203 if isinstance(obj, Message) and obj.pluralizable: 

204 return obj.id[0], obj.context or '' 

205 return obj.id, obj.context or '' 

206 

207 return _cmp(values_to_compare(self), values_to_compare(other)) 

208 

209 def __gt__(self, other: object) -> bool: 

210 return self.__cmp__(other) > 0 

211 

212 def __lt__(self, other: object) -> bool: 

213 return self.__cmp__(other) < 0 

214 

215 def __ge__(self, other: object) -> bool: 

216 return self.__cmp__(other) >= 0 

217 

218 def __le__(self, other: object) -> bool: 

219 return self.__cmp__(other) <= 0 

220 

221 def __eq__(self, other: object) -> bool: 

222 return self.__cmp__(other) == 0 

223 

224 def __ne__(self, other: object) -> bool: 

225 return self.__cmp__(other) != 0 

226 

227 def is_identical(self, other: Message) -> bool: 

228 """Checks whether messages are identical, taking into account all 

229 properties. 

230 """ 

231 assert isinstance(other, Message) 

232 return self.__dict__ == other.__dict__ 

233 

234 def clone(self) -> Message: 

235 return Message( 

236 id=copy(self.id), 

237 string=copy(self.string), 

238 locations=copy(self.locations), 

239 flags=copy(self.flags), 

240 auto_comments=copy(self.auto_comments), 

241 user_comments=copy(self.user_comments), 

242 previous_id=copy(self.previous_id), 

243 lineno=self.lineno, # immutable (str/None) 

244 context=self.context, # immutable (str/None) 

245 ) 

246 

247 def check(self, catalog: Catalog | None = None) -> list[TranslationError]: 

248 """Run various validation checks on the message. Some validations 

249 are only performed if the catalog is provided. This method returns 

250 a sequence of `TranslationError` objects. 

251 

252 :rtype: ``iterator`` 

253 :param catalog: A catalog instance that is passed to the checkers 

254 :see: `Catalog.check` for a way to perform checks for all messages 

255 in a catalog. 

256 """ 

257 from babel.messages.checkers import checkers 

258 

259 errors: list[TranslationError] = [] 

260 for checker in checkers: 

261 try: 

262 checker(catalog, self) 

263 except TranslationError as e: 

264 errors.append(e) 

265 return errors 

266 

267 @property 

268 def fuzzy(self) -> bool: 

269 """Whether the translation is fuzzy. 

270 

271 >>> Message('foo').fuzzy 

272 False 

273 >>> msg = Message('foo', 'foo', flags=['fuzzy']) 

274 >>> msg.fuzzy 

275 True 

276 >>> msg 

277 <Message 'foo' (flags: ['fuzzy'])> 

278 """ 

279 return 'fuzzy' in self.flags 

280 

281 @property 

282 def pluralizable(self) -> bool: 

283 """Whether the message is plurizable. 

284 

285 >>> Message('foo').pluralizable 

286 False 

287 >>> Message(('foo', 'bar')).pluralizable 

288 True 

289 """ 

290 return isinstance(self.id, (list, tuple)) 

291 

292 @property 

293 def python_format(self) -> bool: 

294 """Whether the message contains Python-style parameters. 

295 

296 >>> Message('foo %(name)s bar').python_format 

297 True 

298 >>> Message(('foo %(name)s', 'foo %(name)s')).python_format 

299 True 

300 """ 

301 ids = self.id 

302 if isinstance(ids, (list, tuple)): 

303 for id in ids: # Explicit loop for performance reasons. 

304 if PYTHON_FORMAT.search(id): 

305 return True 

306 return False 

307 return bool(PYTHON_FORMAT.search(ids)) 

308 

309 @property 

310 def python_brace_format(self) -> bool: 

311 """Whether the message contains Python f-string parameters. 

312 

313 >>> Message('Hello, {name}!').python_brace_format 

314 True 

315 >>> Message(('One apple', '{count} apples')).python_brace_format 

316 True 

317 """ 

318 ids = self.id 

319 if isinstance(ids, (list, tuple)): 

320 for id in ids: # Explicit loop for performance reasons. 

321 if _has_python_brace_format(id): 

322 return True 

323 return False 

324 return _has_python_brace_format(ids) 

325 

326 

327class TranslationError(Exception): 

328 """Exception thrown by translation checkers when invalid message 

329 translations are encountered.""" 

330 

331 

332DEFAULT_HEADER = """\ 

333# Translations template for PROJECT. 

334# Copyright (C) YEAR ORGANIZATION 

335# This file is distributed under the same license as the PROJECT project. 

336# FIRST AUTHOR <EMAIL@ADDRESS>, YEAR. 

337#""" 

338 

339 

340def parse_separated_header(value: str) -> dict[str, str]: 

341 # Adapted from https://peps.python.org/pep-0594/#cgi 

342 from email.message import Message 

343 

344 m = Message() 

345 m['content-type'] = value 

346 return dict(m.get_params()) 

347 

348 

349def _force_text(s: str | bytes, encoding: str = 'utf-8', errors: str = 'strict') -> str: 

350 if isinstance(s, str): 

351 return s 

352 if isinstance(s, bytes): 

353 return s.decode(encoding, errors) 

354 return str(s) 

355 

356 

357class ConflictInfo(TypedDict): 

358 message: Message 

359 filename: str 

360 project: str 

361 version: str 

362 

363 

364class Catalog: 

365 """Representation of a message catalog.""" 

366 

367 def __init__( 

368 self, 

369 locale: Locale | str | None = None, 

370 domain: str | None = None, 

371 header_comment: str | None = DEFAULT_HEADER, 

372 project: str | None = None, 

373 version: str | None = None, 

374 copyright_holder: str | None = None, 

375 msgid_bugs_address: str | None = None, 

376 creation_date: datetime.datetime | str | None = None, 

377 revision_date: datetime.datetime | datetime.time | float | str | None = None, 

378 last_translator: str | None = None, 

379 language_team: str | None = None, 

380 charset: str | None = None, 

381 fuzzy: bool = True, 

382 ) -> None: 

383 """Initialize the catalog object. 

384 

385 :param locale: the locale identifier or `Locale` object, or `None` 

386 if the catalog is not bound to a locale (which basically 

387 means it's a template) 

388 :param domain: the message domain 

389 :param header_comment: the header comment as string, or `None` for the 

390 default header 

391 :param project: the project's name 

392 :param version: the project's version 

393 :param copyright_holder: the copyright holder of the catalog 

394 :param msgid_bugs_address: the email address or URL to submit bug 

395 reports to 

396 :param creation_date: the date the catalog was created 

397 :param revision_date: the date the catalog was revised 

398 :param last_translator: the name and email of the last translator 

399 :param language_team: the name and email of the language team 

400 :param charset: the encoding to use in the output (defaults to utf-8) 

401 :param fuzzy: the fuzzy bit on the catalog header 

402 """ 

403 self.domain = domain 

404 self.locale = locale 

405 self._header_comment = header_comment 

406 self._messages: dict[str | tuple[str, str], Message] = {} 

407 self._conflicts: dict[str | tuple[str, str], list[ConflictInfo]] = defaultdict(list) 

408 

409 self.project = project or 'PROJECT' 

410 self.version = version or 'VERSION' 

411 self.copyright_holder = copyright_holder or 'ORGANIZATION' 

412 self.msgid_bugs_address = msgid_bugs_address or 'EMAIL@ADDRESS' 

413 

414 self.last_translator = last_translator or 'FULL NAME <EMAIL@ADDRESS>' 

415 """Name and email address of the last translator.""" 

416 self.language_team = language_team or 'LANGUAGE <LL@li.org>' 

417 """Name and email address of the language team.""" 

418 

419 self.charset = charset or 'utf-8' 

420 

421 if creation_date is None: 

422 creation_date = datetime.datetime.now(LOCALTZ) 

423 elif isinstance(creation_date, datetime.datetime) and not creation_date.tzinfo: 

424 creation_date = creation_date.replace(tzinfo=LOCALTZ) 

425 self.creation_date = creation_date 

426 if revision_date is None: 

427 revision_date = 'YEAR-MO-DA HO:MI+ZONE' 

428 elif isinstance(revision_date, datetime.datetime) and not revision_date.tzinfo: 

429 revision_date = revision_date.replace(tzinfo=LOCALTZ) 

430 self.revision_date = revision_date 

431 self.fuzzy = fuzzy 

432 

433 # Dictionary of obsolete messages 

434 self.obsolete: dict[str | tuple[str, str], Message] = {} 

435 self._num_plurals = None 

436 self._plural_expr = None 

437 

438 def _set_locale(self, locale: Locale | str | None) -> None: 

439 if locale is None: 

440 self._locale_identifier = None 

441 self._locale = None 

442 return 

443 

444 if isinstance(locale, Locale): 

445 self._locale_identifier = str(locale) 

446 self._locale = locale 

447 return 

448 

449 if isinstance(locale, str): 

450 self._locale_identifier = str(locale) 

451 try: 

452 self._locale = Locale.parse(locale) 

453 except UnknownLocaleError: 

454 self._locale = None 

455 return 

456 

457 raise TypeError( 

458 f"`locale` must be a Locale, a locale identifier string, or None; got {locale!r}", 

459 ) 

460 

461 @property 

462 def locale(self) -> Locale | None: 

463 return self._locale 

464 

465 @locale.setter 

466 def locale(self, locale: Locale | str | None) -> None: 

467 self._set_locale(locale) 

468 

469 @property 

470 def locale_identifier(self) -> str | None: 

471 return self._locale_identifier 

472 

473 def _get_header_comment(self) -> str: 

474 comment = self._header_comment 

475 year = datetime.datetime.now(LOCALTZ).strftime('%Y') 

476 if hasattr(self.revision_date, 'strftime'): 

477 year = self.revision_date.strftime('%Y') 

478 comment = ( 

479 comment.replace('PROJECT', self.project) 

480 .replace('VERSION', self.version) 

481 .replace('YEAR', year) 

482 .replace('ORGANIZATION', self.copyright_holder) 

483 ) 

484 locale_name = self.locale.english_name if self.locale else self.locale_identifier 

485 if locale_name: 

486 comment = comment.replace("Translations template", f"{locale_name} translations") 

487 return comment 

488 

489 def _set_header_comment(self, string: str | None) -> None: 

490 self._header_comment = string 

491 

492 @property 

493 def header_comment(self) -> str: 

494 """ 

495 The header comment for the catalog. 

496 

497 >>> catalog = Catalog(project='Foobar', version='1.0', 

498 ... copyright_holder='Foo Company') 

499 >>> print(catalog.header_comment) #doctest: +ELLIPSIS 

500 # Translations template for Foobar. 

501 # Copyright (C) ... Foo Company 

502 # This file is distributed under the same license as the Foobar project. 

503 # FIRST AUTHOR <EMAIL@ADDRESS>, .... 

504 # 

505 

506 The header can also be set from a string. Any known upper-case variables 

507 will be replaced when the header is retrieved again: 

508 

509 >>> catalog = Catalog(project='Foobar', version='1.0', 

510 ... copyright_holder='Foo Company') 

511 >>> catalog.header_comment = '''\\ 

512 ... # The POT for my really cool PROJECT project. 

513 ... # Copyright (C) 1990-2003 ORGANIZATION 

514 ... # This file is distributed under the same license as the PROJECT 

515 ... # project. 

516 ... #''' 

517 >>> print(catalog.header_comment) 

518 # The POT for my really cool Foobar project. 

519 # Copyright (C) 1990-2003 Foo Company 

520 # This file is distributed under the same license as the Foobar 

521 # project. 

522 # 

523 """ 

524 return self._get_header_comment() 

525 

526 @header_comment.setter 

527 def header_comment(self, value: str) -> None: 

528 self._set_header_comment(value) 

529 

530 def _get_mime_headers(self) -> list[tuple[str, str]]: 

531 if isinstance(self.revision_date, (datetime.datetime, datetime.time, int, float)): 

532 revision_date = format_datetime( 

533 self.revision_date, 

534 'yyyy-MM-dd HH:mmZ', 

535 locale='en', 

536 ) 

537 else: 

538 revision_date = self.revision_date 

539 

540 language_team = self.language_team 

541 if self.locale_identifier and 'LANGUAGE' in language_team: 

542 language_team = language_team.replace('LANGUAGE', str(self.locale_identifier)) 

543 

544 headers: list[tuple[str, str]] = [ 

545 ("Project-Id-Version", f"{self.project} {self.version}"), 

546 ('Report-Msgid-Bugs-To', self.msgid_bugs_address), 

547 ('POT-Creation-Date', format_datetime(self.creation_date, 'yyyy-MM-dd HH:mmZ', locale='en')), 

548 ('PO-Revision-Date', revision_date), 

549 ('Last-Translator', self.last_translator), 

550 ] # fmt: skip 

551 if self.locale_identifier: 

552 headers.append(('Language', str(self.locale_identifier))) 

553 headers.append(('Language-Team', language_team)) 

554 if self.locale is not None or self._num_plurals is not None: 

555 # Keep explicit plural forms even when no locale set. 

556 headers.append(('Plural-Forms', self.plural_forms)) 

557 headers += [ 

558 ('MIME-Version', '1.0'), 

559 ("Content-Type", f"text/plain; charset={self.charset}"), 

560 ('Content-Transfer-Encoding', '8bit'), 

561 ("Generated-By", f"Babel {VERSION}\n"), 

562 ] 

563 return headers 

564 

565 def _set_mime_headers(self, headers: Iterable[tuple[str, str]]) -> None: 

566 for name, value in headers: 

567 name = _force_text(name.lower(), encoding=self.charset) 

568 value = _force_text(value, encoding=self.charset) 

569 if name == 'project-id-version': 

570 parts = value.split(' ') 

571 self.project = ' '.join(parts[:-1]) 

572 self.version = parts[-1] 

573 elif name == 'report-msgid-bugs-to': 

574 self.msgid_bugs_address = value 

575 elif name == 'last-translator': 

576 self.last_translator = value 

577 elif name == 'language': 

578 value = value.replace('-', '_') 

579 # The `or None` makes sure that the locale is set to None 

580 # if the header's value is an empty string, which is what 

581 # some tools generate (instead of eliding the empty Language 

582 # header altogether). 

583 self._set_locale(value or None) 

584 elif name == 'language-team': 

585 self.language_team = value 

586 elif name == 'content-type': 

587 params = parse_separated_header(value) 

588 if 'charset' in params: 

589 self.charset = params['charset'].lower() 

590 elif name == 'plural-forms': 

591 params = parse_separated_header(f" ;{value}") 

592 self._num_plurals = int(params.get('nplurals', 2)) 

593 self._plural_expr = params.get('plural', '(n != 1)') 

594 elif name == 'pot-creation-date': 

595 self.creation_date = _parse_datetime_header(value) 

596 elif name == 'po-revision-date': 

597 # Keep the value if it's not the default one 

598 if 'YEAR' not in value: 

599 self.revision_date = _parse_datetime_header(value) 

600 

601 @property 

602 def mime_headers(self) -> list[tuple[str, str]]: 

603 """ 

604 The MIME headers of the catalog, used for the special ``msgid ""`` entry. 

605 

606 The behavior of this property changes slightly depending on whether a locale 

607 is set or not, the latter indicating that the catalog is actually a template 

608 for actual translations. 

609 

610 Here's an example of the output for such a catalog template: 

611 

612 >>> from babel.dates import UTC 

613 >>> from datetime import datetime 

614 >>> created = datetime(1990, 4, 1, 15, 30, tzinfo=UTC) 

615 >>> catalog = Catalog(project='Foobar', version='1.0', 

616 ... creation_date=created) 

617 >>> for name, value in catalog.mime_headers: 

618 ... print('%s: %s' % (name, value)) 

619 Project-Id-Version: Foobar 1.0 

620 Report-Msgid-Bugs-To: EMAIL@ADDRESS 

621 POT-Creation-Date: 1990-04-01 15:30+0000 

622 PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE 

623 Last-Translator: FULL NAME <EMAIL@ADDRESS> 

624 Language-Team: LANGUAGE <LL@li.org> 

625 MIME-Version: 1.0 

626 Content-Type: text/plain; charset=utf-8 

627 Content-Transfer-Encoding: 8bit 

628 Generated-By: Babel ... 

629 

630 And here's an example of the output when the locale is set: 

631 

632 >>> revised = datetime(1990, 8, 3, 12, 0, tzinfo=UTC) 

633 >>> catalog = Catalog(locale='de_DE', project='Foobar', version='1.0', 

634 ... creation_date=created, revision_date=revised, 

635 ... last_translator='John Doe <jd@example.com>', 

636 ... language_team='de_DE <de@example.com>') 

637 >>> for name, value in catalog.mime_headers: 

638 ... print('%s: %s' % (name, value)) 

639 Project-Id-Version: Foobar 1.0 

640 Report-Msgid-Bugs-To: EMAIL@ADDRESS 

641 POT-Creation-Date: 1990-04-01 15:30+0000 

642 PO-Revision-Date: 1990-08-03 12:00+0000 

643 Last-Translator: John Doe <jd@example.com> 

644 Language: de_DE 

645 Language-Team: de_DE <de@example.com> 

646 Plural-Forms: nplurals=2; plural=(n != 1); 

647 MIME-Version: 1.0 

648 Content-Type: text/plain; charset=utf-8 

649 Content-Transfer-Encoding: 8bit 

650 Generated-By: Babel ... 

651 """ 

652 return self._get_mime_headers() 

653 

654 @mime_headers.setter 

655 def mime_headers(self, value: Iterable[tuple[str, str]]) -> None: 

656 self._set_mime_headers(value) 

657 

658 @property 

659 def num_plurals(self) -> int: 

660 """The number of plurals used by the catalog or locale. 

661 

662 >>> Catalog(locale='en').num_plurals 

663 2 

664 >>> Catalog(locale='ga').num_plurals 

665 5 

666 """ 

667 if self._num_plurals is not None: 

668 return self._num_plurals 

669 if self.locale: 

670 return get_plural(self.locale)[0] 

671 return 2 

672 

673 @property 

674 def plural_expr(self) -> str: 

675 """The plural expression used by the catalog or locale. 

676 

677 >>> Catalog(locale='en').plural_expr 

678 '(n != 1)' 

679 >>> Catalog(locale='ga').plural_expr 

680 '(n == 1 ? 0 : n == 2 ? 1 : n >= 3 && n <= 6 ? 2 : n >= 7 && n <= 10 ? 3 : 4)' 

681 >>> Catalog(locale='ding').plural_expr # unknown locale 

682 '(n != 1)' 

683 """ 

684 if self._plural_expr is not None: 

685 return self._plural_expr 

686 if self.locale: 

687 return get_plural(self.locale)[1] 

688 return '(n != 1)' 

689 

690 @property 

691 def plural_forms(self) -> str: 

692 """Return the plural forms declaration for the locale. 

693 

694 >>> Catalog(locale='en').plural_forms 

695 'nplurals=2; plural=(n != 1);' 

696 >>> Catalog(locale='pt_BR').plural_forms 

697 'nplurals=3; plural=(n == 0 || n == 1) ? 0 : n != 0 && n % 1000000 == 0 ? 1 : 2;' 

698 """ 

699 return f"nplurals={self.num_plurals}; plural={self.plural_expr};" 

700 

701 def __contains__(self, id: _MessageID) -> bool: 

702 """Return whether the catalog has a message with the specified ID.""" 

703 return self._key_for(id) in self._messages 

704 

705 def __len__(self) -> int: 

706 """The number of messages in the catalog. 

707 

708 This does not include the special ``msgid ""`` entry.""" 

709 return len(self._messages) 

710 

711 def __iter__(self) -> Iterator[Message]: 

712 """Iterates through all the entries in the catalog, in the order they 

713 were added, yielding a `Message` object for every entry. 

714 

715 :rtype: ``iterator``""" 

716 buf = [] 

717 for name, value in self.mime_headers: 

718 buf.append(f"{name}: {value}") 

719 flags = set() 

720 if self.fuzzy: 

721 flags |= {'fuzzy'} 

722 yield Message('', '\n'.join(buf), flags=flags) 

723 for key in self._messages: 

724 yield self._messages[key] 

725 

726 def __repr__(self) -> str: 

727 locale = '' 

728 if self.locale: 

729 locale = f" {self.locale}" 

730 return f"<{type(self).__name__} {self.domain!r}{locale}>" 

731 

732 def __delitem__(self, id: _MessageID) -> None: 

733 """Delete the message with the specified ID.""" 

734 self.delete(id) 

735 

736 def __getitem__(self, id: _MessageID) -> Message: 

737 """Return the message with the specified ID. 

738 

739 :param id: the message ID 

740 """ 

741 return self.get(id) 

742 

743 def __setitem__(self, id: _MessageID, message: Message) -> None: 

744 """Add or update the message with the specified ID. 

745 

746 >>> catalog = Catalog() 

747 >>> catalog['foo'] = Message('foo') 

748 >>> catalog['foo'] 

749 <Message 'foo' (flags: [])> 

750 

751 If a message with that ID is already in the catalog, it is updated 

752 to include the locations and flags of the new message. 

753 

754 >>> catalog = Catalog() 

755 >>> catalog['foo'] = Message('foo', locations=[('main.py', 1)]) 

756 >>> catalog['foo'].locations 

757 [('main.py', 1)] 

758 >>> catalog['foo'] = Message('foo', locations=[('utils.py', 5)]) 

759 >>> catalog['foo'].locations 

760 [('main.py', 1), ('utils.py', 5)] 

761 

762 :param id: the message ID 

763 :param message: the `Message` object 

764 """ 

765 assert isinstance(message, Message), 'expected a Message object' 

766 key = self._key_for(id, message.context) 

767 current = self._messages.get(key) 

768 if current: 

769 if message.pluralizable and not current.pluralizable: 

770 # The new message adds pluralization 

771 current.id = message.id 

772 current.string = message.string 

773 current.locations = list(dict.fromkeys([*current.locations, *message.locations])) 

774 current.auto_comments = list(dict.fromkeys([*current.auto_comments, *message.auto_comments])) # fmt:skip 

775 current.user_comments = list(dict.fromkeys([*current.user_comments, *message.user_comments])) # fmt:skip 

776 current.flags |= message.flags 

777 elif id == '': 

778 # special treatment for the header message 

779 self.mime_headers = message_from_string(message.string).items() 

780 self.header_comment = "\n".join(f"# {c}".rstrip() for c in message.user_comments) 

781 self.fuzzy = message.fuzzy 

782 else: 

783 if isinstance(id, (list, tuple)): 

784 assert isinstance(message.string, (list, tuple)), ( 

785 f"Expected sequence but got {type(message.string)}" 

786 ) 

787 self._messages[key] = message 

788 

789 def add_conflict(self, message: Message, filename: str, project: str, version: str) -> None: 

790 """Record a conflicting translation for a message. 

791 

792 When the same message ID has different translations across input files, 

793 the conflicting entry is stored and the message is marked as fuzzy in 

794 the output catalog. 

795 

796 :param message: the conflicting :class:`Message` object 

797 :param filename: the basename of the file where the conflict originates 

798 :param project: the project name of the conflicting file 

799 :param version: the project version of the conflicting file 

800 """ 

801 key = self._key_for(message.id, message.context) 

802 self._conflicts[key].append({ 

803 'message': message, 

804 'filename': filename, 

805 'project': project, 

806 'version': version, 

807 }) 

808 

809 def get_conflicts(self, id: _MessageID, context: str | None = None) -> list[ConflictInfo]: 

810 """Return all recorded conflicts for a message ID. 

811 

812 :param id: the message ID to look up conflicts for 

813 :param context: optional message context (msgctxt) 

814 :return: list of :class:`ConflictInfo` dicts, or an empty list if none 

815 """ 

816 key = self._key_for(id, context) 

817 return self._conflicts.get(key, []) 

818 

819 def add( 

820 self, 

821 id: _MessageID, 

822 string: _MessageID | None = None, 

823 locations: Iterable[tuple[str, int]] = (), 

824 flags: Iterable[str] = (), 

825 auto_comments: Iterable[str] = (), 

826 user_comments: Iterable[str] = (), 

827 previous_id: _MessageID = (), 

828 lineno: int | None = None, 

829 context: str | None = None, 

830 ) -> Message: 

831 """Add or update the message with the specified ID. 

832 

833 >>> catalog = Catalog() 

834 >>> catalog.add('foo') 

835 <Message ...> 

836 >>> catalog['foo'] 

837 <Message 'foo' (flags: [])> 

838 

839 This method simply constructs a `Message` object with the given 

840 arguments and invokes `__setitem__` with that object. 

841 

842 :param id: the message ID, or a ``(singular, plural)`` tuple for 

843 pluralizable messages 

844 :param string: the translated message string, or a 

845 ``(singular, plural)`` tuple for pluralizable messages 

846 :param locations: a sequence of ``(filename, lineno)`` tuples 

847 :param flags: a set or sequence of flags 

848 :param auto_comments: a sequence of automatic comments 

849 :param user_comments: a sequence of user comments 

850 :param previous_id: the previous message ID, or a ``(singular, plural)`` 

851 tuple for pluralizable messages 

852 :param lineno: the line number on which the msgid line was found in the 

853 PO file, if any 

854 :param context: the message context 

855 """ 

856 message = Message( 

857 id, 

858 string, 

859 list(locations), 

860 flags, 

861 auto_comments, 

862 user_comments, 

863 previous_id, 

864 lineno=lineno, 

865 context=context, 

866 ) 

867 self[id] = message 

868 return message 

869 

870 def check(self) -> Iterable[tuple[Message, list[TranslationError]]]: 

871 """Run various validation checks on the translations in the catalog. 

872 

873 For every message which fails validation, this method yield a 

874 ``(message, errors)`` tuple, where ``message`` is the `Message` object 

875 and ``errors`` is a sequence of `TranslationError` objects. 

876 

877 :rtype: ``generator`` of ``(message, errors)`` 

878 """ 

879 for message in self._messages.values(): 

880 errors = message.check(catalog=self) 

881 if errors: 

882 yield message, errors 

883 

884 def get(self, id: _MessageID, context: str | None = None) -> Message | None: 

885 """Return the message with the specified ID and context. 

886 

887 :param id: the message ID 

888 :param context: the message context, or ``None`` for no context 

889 """ 

890 return self._messages.get(self._key_for(id, context)) 

891 

892 def delete(self, id: _MessageID, context: str | None = None) -> None: 

893 """Delete the message with the specified ID and context. 

894 

895 :param id: the message ID 

896 :param context: the message context, or ``None`` for no context 

897 """ 

898 key = self._key_for(id, context) 

899 if key in self._messages: 

900 del self._messages[key] 

901 

902 def update( 

903 self, 

904 template: Catalog, 

905 no_fuzzy_matching: bool = False, 

906 update_header_comment: bool = False, 

907 keep_user_comments: bool = True, 

908 update_creation_date: bool = True, 

909 ) -> None: 

910 """Update the catalog based on the given template catalog. 

911 

912 >>> from babel.messages import Catalog 

913 >>> template = Catalog() 

914 >>> template.add('green', locations=[('main.py', 99)]) 

915 <Message ...> 

916 >>> template.add('blue', locations=[('main.py', 100)]) 

917 <Message ...> 

918 >>> template.add(('salad', 'salads'), locations=[('util.py', 42)]) 

919 <Message ...> 

920 >>> catalog = Catalog(locale='de_DE') 

921 >>> catalog.add('blue', 'blau', locations=[('main.py', 98)]) 

922 <Message ...> 

923 >>> catalog.add('head', 'Kopf', locations=[('util.py', 33)]) 

924 <Message ...> 

925 >>> catalog.add(('salad', 'salads'), ('Salat', 'Salate'), 

926 ... locations=[('util.py', 38)]) 

927 <Message ...> 

928 

929 >>> catalog.update(template) 

930 >>> len(catalog) 

931 3 

932 

933 >>> msg1 = catalog['green'] 

934 >>> msg1.string 

935 >>> msg1.locations 

936 [('main.py', 99)] 

937 

938 >>> msg2 = catalog['blue'] 

939 >>> msg2.string 

940 'blau' 

941 >>> msg2.locations 

942 [('main.py', 100)] 

943 

944 >>> msg3 = catalog['salad'] 

945 >>> msg3.string 

946 ('Salat', 'Salate') 

947 >>> msg3.locations 

948 [('util.py', 42)] 

949 

950 Messages that are in the catalog but not in the template are removed 

951 from the main collection, but can still be accessed via the `obsolete` 

952 member: 

953 

954 >>> 'head' in catalog 

955 False 

956 >>> list(catalog.obsolete.values()) 

957 [<Message 'head' (flags: [])>] 

958 

959 :param template: the reference catalog, usually read from a POT file 

960 :param no_fuzzy_matching: whether to use fuzzy matching of message IDs 

961 :param update_header_comment: whether to copy the header comment from the template 

962 :param keep_user_comments: whether to keep user comments from the old catalog 

963 :param update_creation_date: whether to copy the creation date from the template 

964 """ 

965 messages = self._messages 

966 remaining = messages.copy() 

967 self._messages = {} 

968 

969 # Prepare for fuzzy matching 

970 fuzzy_candidates = {} 

971 if not no_fuzzy_matching: 

972 for msgid in messages: 

973 if msgid and messages[msgid].string: 

974 key = self._key_for(msgid) 

975 ctxt = messages[msgid].context 

976 fuzzy_candidates[self._to_fuzzy_match_key(key)] = (key, ctxt) 

977 fuzzy_matches = set() 

978 

979 def _merge( 

980 message: Message, 

981 oldkey: tuple[str, str] | str, 

982 newkey: tuple[str, str] | str, 

983 ) -> None: 

984 message = message.clone() 

985 fuzzy = False 

986 if oldkey != newkey: 

987 fuzzy = True 

988 fuzzy_matches.add(oldkey) 

989 oldmsg = messages.get(oldkey) 

990 assert oldmsg is not None 

991 if isinstance(oldmsg.id, str): 

992 message.previous_id = [oldmsg.id] 

993 else: 

994 message.previous_id = list(oldmsg.id) 

995 else: 

996 oldmsg = remaining.pop(oldkey, None) 

997 assert oldmsg is not None 

998 message.string = oldmsg.string 

999 

1000 if keep_user_comments and oldmsg.user_comments: 

1001 message.user_comments = list(dict.fromkeys(oldmsg.user_comments)) 

1002 

1003 if isinstance(message.id, (list, tuple)): 

1004 if not isinstance(message.string, (list, tuple)): 

1005 fuzzy = True 

1006 message.string = tuple( 

1007 [message.string] + ([''] * (len(message.id) - 1)), 

1008 ) 

1009 elif len(message.string) != self.num_plurals: 

1010 fuzzy = True 

1011 message.string = tuple(message.string[: len(oldmsg.string)]) 

1012 elif isinstance(message.string, (list, tuple)): 

1013 fuzzy = True 

1014 message.string = message.string[0] 

1015 message.flags |= oldmsg.flags 

1016 if fuzzy: 

1017 message.flags |= {'fuzzy'} 

1018 self[message.id] = message 

1019 

1020 for message in template: 

1021 if message.id: 

1022 key = self._key_for(message.id, message.context) 

1023 if key in messages: 

1024 _merge(message, key, key) 

1025 else: 

1026 if not no_fuzzy_matching: 

1027 # do some fuzzy matching with difflib 

1028 matches = get_close_matches( 

1029 self._to_fuzzy_match_key(key), 

1030 fuzzy_candidates.keys(), 

1031 1, 

1032 ) 

1033 if matches: 

1034 modified_key = matches[0] 

1035 newkey, newctxt = fuzzy_candidates[modified_key] 

1036 if newctxt is not None: 

1037 newkey = newkey, newctxt 

1038 _merge(message, newkey, key) 

1039 continue 

1040 

1041 self[message.id] = message 

1042 

1043 for msgid in remaining: 

1044 if no_fuzzy_matching or msgid not in fuzzy_matches: 

1045 self.obsolete[msgid] = remaining[msgid] 

1046 

1047 if update_header_comment: 

1048 # Allow the updated catalog's header to be rewritten based on the 

1049 # template's header 

1050 self.header_comment = template.header_comment 

1051 

1052 # Make updated catalog's POT-Creation-Date equal to the template 

1053 # used to update the catalog 

1054 if update_creation_date: 

1055 self.creation_date = template.creation_date 

1056 

1057 def _to_fuzzy_match_key(self, key: tuple[str, str] | str) -> str: 

1058 """Converts a message key to a string suitable for fuzzy matching.""" 

1059 if isinstance(key, tuple): 

1060 matchkey = key[0] # just the msgid, no context 

1061 else: 

1062 matchkey = key 

1063 return matchkey.lower().strip() 

1064 

1065 def _key_for( 

1066 self, 

1067 id: _MessageID, 

1068 context: str | None = None, 

1069 ) -> tuple[str, str] | str: 

1070 """The key for a message is just the singular ID even for pluralizable 

1071 messages, but is a ``(msgid, msgctxt)`` tuple for context-specific 

1072 messages. 

1073 """ 

1074 key = id 

1075 if isinstance(key, (list, tuple)): 

1076 key = id[0] 

1077 if context is not None: 

1078 key = (key, context) 

1079 return key 

1080 

1081 def is_identical(self, other: Catalog) -> bool: 

1082 """Checks if catalogs are identical, taking into account messages and 

1083 headers. 

1084 """ 

1085 assert isinstance(other, Catalog) 

1086 for key in self._messages.keys() | other._messages.keys(): 

1087 message_1 = self.get(key) 

1088 message_2 = other.get(key) 

1089 if message_1 is None or message_2 is None or not message_1.is_identical(message_2): 

1090 return False 

1091 return dict(self.mime_headers) == dict(other.mime_headers)