Coverage for /pythoncovmergedfiles/medio/medio/usr/local/lib/python3.11/site-packages/pyparsing/helpers.py: 17%
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
Shortcuts on this page
r m x toggle line displays
j k next/prev highlighted chunk
0 (zero) top of page
1 (one) first highlighted chunk
1# helpers.py
2from enum import Enum, auto
3import html.entities
4import operator
5import re
6import sys
7import typing
9from . import __diag__
10from .core import *
11from .util import (
12 _bslash,
13 _flatten,
14 _escape_regex_range_chars,
15 make_compressed_re,
16 replaced_by_pep8,
17)
20def _suppression(expr: Union[ParserElement, str]) -> ParserElement:
21 # internal helper to avoid wrapping Suppress inside another Suppress
22 if isinstance(expr, Suppress):
23 return expr
24 return Suppress(expr)
27#
28# global helpers
29#
30def counted_array(
31 expr: ParserElement, int_expr: typing.Optional[ParserElement] = None, **kwargs
32) -> ParserElement:
33 """Helper to define a counted list of expressions.
35 This helper defines a pattern of the form::
37 integer expr expr expr...
39 where the leading integer tells how many expr expressions follow.
40 The matched tokens returns the array of expr tokens as a list - the
41 leading count token is suppressed.
43 If ``int_expr`` is specified, it should be a pyparsing expression
44 that produces an integer value.
46 Examples:
48 .. doctest::
50 >>> counted_array(Word(alphas)).parse_string('2 ab cd ef')
51 ParseResults(['ab', 'cd'], {})
53 - In this parser, the leading integer value is given in binary,
54 '10' indicating that 2 values are in the array:
56 .. doctest::
58 >>> binary_constant = Word('01').set_parse_action(lambda t: int(t[0], 2))
59 >>> counted_array(Word(alphas), int_expr=binary_constant
60 ... ).parse_string('10 ab cd ef')
61 ParseResults(['ab', 'cd'], {})
63 - If other fields must be parsed after the count but before the
64 list items, give the fields results names and they will
65 be preserved in the returned ParseResults:
67 .. doctest::
69 >>> ppc = pyparsing.common
70 >>> count_with_metadata = ppc.integer + Word(alphas)("type")
71 >>> typed_array = counted_array(Word(alphanums),
72 ... int_expr=count_with_metadata)("items")
73 >>> result = typed_array.parse_string("3 bool True True False")
74 >>> print(result.dump())
75 ['True', 'True', 'False']
76 - items: ['True', 'True', 'False']
77 - type: 'bool'
78 """
79 intExpr: typing.Optional[ParserElement] = deprecate_argument(
80 kwargs, "intExpr", None
81 )
83 intExpr = intExpr or int_expr
84 array_expr = Forward()
86 def count_field_parse_action(s, l, t):
87 nonlocal array_expr
88 n = t[0]
89 array_expr <<= (expr * n) if n else Empty()
90 # clear list contents, but keep any named results
91 del t[:]
93 if intExpr is None:
94 intExpr = Word(nums).set_parse_action(lambda t: int(t[0]))
95 else:
96 intExpr = intExpr.copy()
97 intExpr.set_name("arrayLen")
98 intExpr.add_parse_action(count_field_parse_action, call_during_try=True)
99 return (intExpr + array_expr).set_name(f"(len) {expr}...")
102def match_previous_literal(expr: ParserElement) -> ParserElement:
103 """Helper to define an expression that is indirectly defined from
104 the tokens matched in a previous expression, that is, it looks for
105 a 'repeat' of a previous expression. For example::
107 .. testcode::
109 first = Word(nums)
110 second = match_previous_literal(first)
111 match_expr = first + ":" + second
113 will match ``"1:1"``, but not ``"1:2"``. Because this
114 matches a previous literal, will also match the leading
115 ``"1:1"`` in ``"1:10"``. If this is not desired, use
116 :class:`match_previous_expr`. Do *not* use with packrat parsing
117 enabled.
118 """
119 rep = Forward()
121 def copy_token_to_repeater(s, l, t):
122 if not t:
123 rep << Empty()
124 return
126 if len(t) == 1:
127 rep << t[0]
128 return
130 # flatten t tokens
131 tflat = _flatten(t.as_list())
132 rep << And(Literal(tt) for tt in tflat)
134 expr.add_parse_action(copy_token_to_repeater, call_during_try=True)
135 rep.set_name(f"(prev) {expr}")
136 return rep
139def match_previous_expr(expr: ParserElement) -> ParserElement:
140 """Helper to define an expression that is indirectly defined from
141 the tokens matched in a previous expression, that is, it looks for
142 a 'repeat' of a previous expression. For example:
144 .. testcode::
146 first = Word(nums)
147 second = match_previous_expr(first)
148 match_expr = first + ":" + second
150 will match ``"1:1"``, but not ``"1:2"``. Because this
151 matches by expressions, will *not* match the leading ``"1:1"``
152 in ``"1:10"``; the expressions are evaluated first, and then
153 compared, so ``"1"`` is compared with ``"10"``. Do *not* use
154 with packrat parsing enabled.
155 """
156 rep = Forward()
157 e2 = expr.copy()
158 rep <<= e2
160 def copy_token_to_repeater(s, l, t):
161 matchTokens = _flatten(t.as_list())
163 def must_match_these_tokens(s, l, t):
164 theseTokens = _flatten(t.as_list())
165 if theseTokens != matchTokens:
166 raise ParseException(
167 s, l, f"Expected {matchTokens}, found{theseTokens}"
168 )
170 rep.set_parse_action(must_match_these_tokens, call_during_try=True)
172 expr.add_parse_action(copy_token_to_repeater, call_during_try=True)
173 rep.set_name(f"(prev) {expr}")
174 return rep
177def one_of(
178 strs: Union[typing.Iterable[str], str],
179 caseless: bool = False,
180 use_regex: bool = True,
181 as_keyword: bool = False,
182 **kwargs,
183) -> ParserElement:
184 """Helper to quickly define a set of alternative :class:`Literal` s,
185 and makes sure to do longest-first testing when there is a conflict,
186 regardless of the input order, but returns
187 a :class:`MatchFirst` for best performance.
189 :param strs: a string of space-delimited literals, or a collection of
190 string literals
191 :param caseless: treat all literals as caseless
192 :param use_regex: bool - as an optimization, will
193 generate a :class:`Regex` object; otherwise, will generate
194 a :class:`MatchFirst` object (if ``caseless=True`` or
195 ``as_keyword=True``, or if creating a :class:`Regex` raises an exception)
196 :param as_keyword: bool - enforce :class:`Keyword`-style matching on the
197 generated expressions
199 Parameters ``asKeyword`` and ``useRegex`` are retained for pre-PEP8
200 compatibility, but will be removed in a future release.
202 Example:
204 .. testcode::
206 comp_oper = one_of("< = > <= >= !=")
207 var = Word(alphas)
208 number = Word(nums)
209 term = var | number
210 comparison_expr = term + comp_oper + term
211 print(comparison_expr.search_string("B = 12 AA=23 B<=AA AA>12"))
213 prints:
215 .. testoutput::
217 [['B', '=', '12'], ['AA', '=', '23'], ['B', '<=', 'AA'], ['AA', '>', '12']]
218 """
219 useRegex: bool = deprecate_argument(kwargs, "useRegex", True)
220 asKeyword: bool = deprecate_argument(kwargs, "asKeyword", False)
222 asKeyword = asKeyword or as_keyword
223 useRegex = useRegex and use_regex
225 if (
226 isinstance(caseless, str_type)
227 and __diag__.warn_on_multiple_string_args_to_oneof
228 ):
229 warnings.warn(
230 "warn_on_multiple_string_args_to_oneof:"
231 " More than one string argument passed to one_of, pass"
232 " choices as a list or space-delimited string",
233 PyparsingDiagnosticWarning,
234 stacklevel=2,
235 )
237 if caseless:
238 is_equal = lambda a, b: a.upper() == b.upper()
239 masks = lambda a, b: b.upper().startswith(a.upper())
240 else:
241 is_equal = operator.eq
242 masks = lambda a, b: b.startswith(a)
244 symbols: list[str]
245 if isinstance(strs, str_type):
246 strs = typing.cast(str, strs)
247 symbols = strs.split()
248 elif isinstance(strs, Iterable):
249 symbols = list(strs)
250 else:
251 raise TypeError("Invalid argument to one_of, expected string or iterable")
252 if not symbols:
253 return NoMatch()
255 # reorder given symbols to take care to avoid masking longer choices with shorter ones
256 # (but only if the given symbols are not just single characters)
257 i = 0
258 while i < len(symbols) - 1:
259 cur = symbols[i]
260 for j, other in enumerate(symbols[i + 1 :]):
261 if is_equal(other, cur):
262 del symbols[i + j + 1]
263 break
264 if len(other) > len(cur) and masks(cur, other):
265 del symbols[i + j + 1]
266 symbols.insert(i, other)
267 break
268 else:
269 i += 1
271 if useRegex:
272 re_flags: int = re.IGNORECASE if caseless else 0
274 try:
275 if all(len(sym) == 1 for sym in symbols):
276 # symbols are just single characters, create range regex pattern
277 patt = f"[{''.join(_escape_regex_range_chars(sym) for sym in symbols)}]"
278 else:
279 patt = "|".join(re.escape(sym) for sym in symbols)
281 # wrap with \b word break markers if defining as keywords
282 if asKeyword:
283 patt = rf"\b(?:{patt})\b"
285 ret = Regex(patt, flags=re_flags)
286 ret.set_name(" | ".join(repr(s) for s in symbols))
288 if caseless:
289 # add parse action to return symbols as specified, not in random
290 # casing as found in input string
291 symbol_map = {sym.lower(): sym for sym in symbols}
292 ret.add_parse_action(lambda s, l, t: symbol_map[t[0].lower()])
294 return ret
296 except re.error:
297 warnings.warn(
298 "Exception creating Regex for one_of, building MatchFirst",
299 PyparsingDiagnosticWarning,
300 stacklevel=2,
301 )
303 # last resort, just use MatchFirst of Token class corresponding to caseless
304 # and asKeyword settings
305 CASELESS = KEYWORD = True
306 parse_element_class = {
307 (CASELESS, KEYWORD): CaselessKeyword,
308 (CASELESS, not KEYWORD): CaselessLiteral,
309 (not CASELESS, KEYWORD): Keyword,
310 (not CASELESS, not KEYWORD): Literal,
311 }[(caseless, asKeyword)]
312 return MatchFirst(parse_element_class(sym) for sym in symbols).set_name(
313 " | ".join(symbols)
314 )
317def dict_of(key: ParserElement, value: ParserElement) -> Dict:
318 """Helper to easily and clearly define a dictionary by specifying
319 the respective patterns for the key and value. Takes care of
320 defining the :class:`Dict`, :class:`ZeroOrMore`, and
321 :class:`Group` tokens in the proper order. The key pattern
322 can include delimiting markers or punctuation, as long as they are
323 suppressed, thereby leaving the significant key text. The value
324 pattern can include named results, so that the :class:`Dict` results
325 can include named token fields.
327 Example:
329 .. doctest::
331 >>> text = "shape: SQUARE posn: upper left color: light blue texture: burlap"
333 >>> data_word = Word(alphas)
334 >>> label = data_word + FollowedBy(':')
335 >>> attr_expr = (
336 ... label
337 ... + Suppress(':')
338 ... + OneOrMore(data_word, stop_on=label)
339 ... .set_parse_action(' '.join))
340 >>> print(attr_expr[1, ...].parse_string(text).dump())
341 ['shape', 'SQUARE', 'posn', 'upper left', 'color', 'light blue', 'texture', 'burlap']
343 >>> attr_label = label
344 >>> attr_value = Suppress(':') + OneOrMore(data_word, stop_on=label
345 ... ).set_parse_action(' '.join)
347 # similar to Dict, but simpler call format
348 >>> result = dict_of(attr_label, attr_value).parse_string(text)
349 >>> print(result.dump())
350 [['shape', 'SQUARE'], ['posn', 'upper left'], ['color', 'light blue'], ['texture', 'burlap']]
351 - color: 'light blue'
352 - posn: 'upper left'
353 - shape: 'SQUARE'
354 - texture: 'burlap'
355 [0]:
356 ['shape', 'SQUARE']
357 [1]:
358 ['posn', 'upper left']
359 [2]:
360 ['color', 'light blue']
361 [3]:
362 ['texture', 'burlap']
364 >>> print(result['shape'])
365 SQUARE
366 >>> print(result.shape) # object attribute access works too
367 SQUARE
368 >>> print(result.as_dict())
369 {'shape': 'SQUARE', 'posn': 'upper left', 'color': 'light blue', 'texture': 'burlap'}
370 """
371 return Dict(OneOrMore(Group(key + value)))
374def original_text_for(
375 expr: ParserElement, as_string: bool = True, **kwargs
376) -> ParserElement:
377 """Helper to return the original, untokenized text for a given
378 expression. Useful to restore the parsed fields of an HTML start
379 tag into the raw tag text itself, or to revert separate tokens with
380 intervening whitespace back to the original matching input text. By
381 default, returns a string containing the original parsed text.
383 If the optional ``as_string`` argument is passed as
384 ``False``, then the return value is
385 a :class:`ParseResults` containing any results names that
386 were originally matched, and a single token containing the original
387 matched text from the input string. So if the expression passed to
388 :class:`original_text_for` contains expressions with defined
389 results names, you must set ``as_string`` to ``False`` if you
390 want to preserve those results name values.
392 The ``asString`` pre-PEP8 argument is retained for compatibility,
393 but will be removed in a future release.
395 Example:
397 .. testcode::
399 src = "this is test <b> bold <i>text</i> </b> normal text "
400 for tag in ("b", "i"):
401 opener, closer = make_html_tags(tag)
402 patt = original_text_for(opener + ... + closer)
403 print(patt.search_string(src)[0])
405 prints:
407 .. testoutput::
409 ['<b> bold <i>text</i> </b>']
410 ['<i>text</i>']
411 """
412 asString: bool = deprecate_argument(kwargs, "asString", True)
414 asString = asString and as_string
416 locMarker = Empty().set_parse_action(lambda s, loc, t: loc)
417 endlocMarker = locMarker.copy()
418 endlocMarker.callPreparse = False
419 matchExpr = locMarker("_original_start") + expr + endlocMarker("_original_end")
420 if asString:
421 extractText = lambda s, l, t: s[t._original_start : t._original_end]
422 else:
424 def extractText(s, l, t):
425 t[:] = [s[t.pop("_original_start") : t.pop("_original_end")]]
427 matchExpr.set_parse_action(extractText)
428 matchExpr.ignoreExprs = expr.ignoreExprs
429 matchExpr.suppress_warning(Diagnostics.warn_ungrouped_named_tokens_in_collection)
430 return matchExpr
433def ungroup(expr: ParserElement) -> ParserElement:
434 """Helper to undo pyparsing's default grouping of And expressions,
435 even if all but one are non-empty.
436 """
437 return TokenConverter(expr).add_parse_action(lambda t: t[0])
440def locatedExpr(expr: ParserElement) -> ParserElement:
441 """
442 .. deprecated:: 3.0.0
443 Use the :class:`Located` class instead. Note that `Located`
444 returns results with one less grouping level.
446 Helper to decorate a returned token with its starting and ending
447 locations in the input string.
449 This helper adds the following results names:
451 - ``locn_start`` - location where matched expression begins
452 - ``locn_end`` - location where matched expression ends
453 - ``value`` - the actual parsed results
455 Be careful if the input text contains ``<TAB>`` characters, you
456 may want to call :meth:`ParserElement.parse_with_tabs`
457 """
458 warnings.warn(
459 f"{'locatedExpr'!r} deprecated - use {'Located'!r}",
460 PyparsingDeprecationWarning,
461 stacklevel=2,
462 )
464 locator = Empty().set_parse_action(lambda ss, ll, tt: ll)
465 return Group(
466 locator("locn_start")
467 + expr("value")
468 + locator.copy().leave_whitespace()("locn_end")
469 )
472# define special default value to permit None as a significant value for
473# ignore_expr
474_NO_IGNORE_EXPR_GIVEN = NoMatch()
477class _NestedExpr(ParseElementEnhance):
478 """Helper method for defining nested lists enclosed in opening and
479 closing delimiters (``"("`` and ``")"`` are the default).
481 :param opener: str - opening character for a nested list
482 (default= ``"("``); can also be a pyparsing expression
484 :param closer: str - closing character for a nested list
485 (default= ``")"``); can also be a pyparsing expression
487 :param content: expression for items within the nested lists
489 :param ignore_expr: expression for ignoring opening and closing delimiters
490 (default = :class:`quoted_string`)
492 Parameter ``ignoreExpr`` is retained for compatibility
493 but will be removed in a future release.
495 If an expression is not provided for the content argument, the
496 nested expression will capture all whitespace-delimited content
497 between delimiters as a list of separate values.
499 Use the ``ignore_expr`` argument to define expressions that may
500 contain opening or closing characters that should not be treated as
501 opening or closing characters for nesting, such as quoted_string or
502 a comment expression. Specify multiple expressions using an
503 :class:`Or` or :class:`MatchFirst`. The default is
504 :class:`quoted_string`, but if no expressions are to be ignored, then
505 pass ``None`` for this argument.
507 Example:
509 .. testcode::
511 data_type = one_of("void int short long char float double")
512 decl_data_type = Combine(data_type + Opt(Word('*')))
513 ident = Word(alphas+'_', alphanums+'_')
514 number = pyparsing_common.number
515 arg = Group(decl_data_type + ident)
516 LPAR, RPAR = map(Suppress, "()")
518 code_body = nested_expr('{', '}', ignore_expr=(quoted_string | c_style_comment))
520 c_function = (decl_data_type("type")
521 + ident("name")
522 + LPAR + Opt(DelimitedList(arg), [])("args") + RPAR
523 + code_body("body"))
524 c_function.ignore(c_style_comment)
526 source_code = '''
527 int is_odd(int x) {
528 return (x%2);
529 }
531 int dec_to_hex(char hchar) {
532 if (hchar >= '0' && hchar <= '9') {
533 return (ord(hchar)-ord('0'));
534 } else {
535 return (10+ord(hchar)-ord('A'));
536 }
537 }
538 '''
539 for func in c_function.search_string(source_code):
540 print(f"{func.name} ({func.type}) args: {func.args}")
543 prints:
545 .. testoutput::
547 is_odd (int) args: [['int', 'x']]
548 dec_to_hex (int) args: [['char', 'hchar']]
549 """
551 def __init__(
552 self,
553 opener: Union[str, ParserElement] = "(",
554 closer: Union[str, ParserElement] = ")",
555 content: typing.Optional[ParserElement] = None,
556 ignore_expr: typing.Optional[ParserElement] = _NO_IGNORE_EXPR_GIVEN,
557 **kwargs,
558 ):
559 ignoreExpr = deprecate_argument(kwargs, "ignoreExpr", _NO_IGNORE_EXPR_GIVEN)
560 if ignoreExpr != ignore_expr:
561 ignoreExpr = (
562 ignore_expr if ignoreExpr is _NO_IGNORE_EXPR_GIVEN else ignoreExpr
563 )
564 if ignoreExpr is _NO_IGNORE_EXPR_GIVEN:
565 ignoreExpr = quoted_string()
567 if opener == closer:
568 raise ValueError("opening and closing strings cannot be the same")
570 original_content = content
571 if content is None:
572 if isinstance(opener, str_type) and isinstance(closer, str_type):
573 opener_str = str(opener)
574 closer_str = str(closer)
575 if len(opener_str) == 1 and len(closer_str) == 1:
576 if ignoreExpr is not None:
577 content = Combine(
578 OneOrMore(
579 ~ignoreExpr
580 + CharsNotIn(
581 opener_str
582 + closer_str
583 + ParserElement.DEFAULT_WHITE_CHARS,
584 exact=1,
585 )
586 )
587 )
588 else:
589 content = Combine(
590 Empty()
591 + CharsNotIn(
592 opener_str
593 + closer_str
594 + ParserElement.DEFAULT_WHITE_CHARS
595 )
596 )
597 else:
598 if ignoreExpr is not None:
599 content = Combine(
600 OneOrMore(
601 ~ignoreExpr
602 + ~Literal(opener_str)
603 + ~Literal(closer_str)
604 + CharsNotIn(ParserElement.DEFAULT_WHITE_CHARS, exact=1)
605 )
606 )
607 else:
608 content = Combine(
609 OneOrMore(
610 ~Literal(opener_str)
611 + ~Literal(closer_str)
612 + CharsNotIn(ParserElement.DEFAULT_WHITE_CHARS, exact=1)
613 )
614 )
615 else:
616 raise ValueError(
617 "opening and closing arguments must be strings if no content expression is given"
618 )
620 if ParserElement.DEFAULT_WHITE_CHARS:
621 content.set_parse_action(
622 lambda t: t[0].strip(ParserElement.DEFAULT_WHITE_CHARS)
623 )
625 super().__init__(content, savelist=True)
626 self.opener = _suppression(opener)
627 self.closer = _suppression(closer)
628 self.opener_raw = opener
629 self.closer_raw = closer
630 self.content = content
631 self.ignore_expr = ignoreExpr
632 self.saveAsList = True
633 self.errmsg = None
634 self.original_content = original_content
636 def _generateDefaultName(self) -> str:
637 if self.original_content is None:
638 return f"nested {self.opener_raw}{self.closer_raw} expression"
639 else:
640 return f"nested {self.opener_raw}{self.original_content}{self.closer_raw} expression"
642 def parseImpl(self, instring, loc, do_actions=True):
643 loc, _ = self.opener._parse(instring, loc, do_actions=do_actions)
644 stack = [[]]
645 while stack:
646 # 1. try ignore_expr
647 if self.ignore_expr is not None:
648 try:
649 loc, toks = self.ignore_expr._parse(
650 instring, loc, do_actions=do_actions
651 )
652 stack[-1].extend(toks)
653 continue
654 except ParseException:
655 pass
656 # 2. try opener
657 try:
658 loc, _ = self.opener._parse(instring, loc, do_actions=do_actions)
659 stack.append([])
660 continue
661 except ParseException:
662 pass
663 # 3. try content
664 try:
665 next_loc, toks = self.content._parse(
666 instring, loc, do_actions=do_actions
667 )
668 if next_loc > loc:
669 loc = next_loc
670 stack[-1].extend(toks)
671 continue
672 except ParseException:
673 pass
674 # 4. try closer
675 try:
676 loc, _ = self.closer._parse(instring, loc, do_actions=do_actions)
677 top = ParseResults(stack.pop())
678 if stack:
679 stack[-1].append(top)
680 else:
681 return loc, ParseResults([top])
682 continue
683 except ParseException:
684 pass
686 raise ParseException(instring, loc, f"Expected {self.closer_raw!r}")
689def nested_expr(
690 opener: Union[str, ParserElement] = "(",
691 closer: Union[str, ParserElement] = ")",
692 content: typing.Optional[ParserElement] = None,
693 ignore_expr: typing.Optional[ParserElement] = _NO_IGNORE_EXPR_GIVEN,
694 **kwargs,
695) -> ParserElement:
696 """Helper method for defining nested lists enclosed in opening and
697 closing delimiters (``"("`` and ``")"`` are the default).
699 :param opener: str - opening character for a nested list
700 (default= ``"("``); can also be a pyparsing expression
702 :param closer: str - closing character for a nested list
703 (default= ``")"``); can also be a pyparsing expression
705 :param content: expression for items within the nested lists
707 :param ignore_expr: expression for ignoring opening and closing delimiters
708 (default = :class:`quoted_string`)
710 Parameter ``ignoreExpr`` is retained for compatibility
711 but will be removed in a future release.
713 If an expression is not provided for the content argument, the
714 nested expression will capture all whitespace-delimited content
715 between delimiters as a list of separate values.
717 Use the ``ignore_expr`` argument to define expressions that may
718 contain opening or closing characters that should not be treated as
719 opening or closing characters for nesting, such as quoted_string or
720 a comment expression. Specify multiple expressions using an
721 :class:`Or` or :class:`MatchFirst`. The default is
722 :class:`quoted_string`, but if no expressions are to be ignored, then
723 pass ``None`` for this argument.
725 Example:
727 .. testcode::
729 data_type = one_of("void int short long char float double")
730 decl_data_type = Combine(data_type + Opt(Word('*')))
731 ident = Word(alphas+'_', alphanums+'_')
732 number = pyparsing_common.number
733 arg = Group(decl_data_type + ident)
734 LPAR, RPAR = map(Suppress, "()")
736 code_body = nested_expr('{', '}', ignore_expr=(quoted_string | c_style_comment))
738 c_function = (decl_data_type("type")
739 + ident("name")
740 + LPAR + Opt(DelimitedList(arg), [])("args") + RPAR
741 + code_body("body"))
742 c_function.ignore(c_style_comment)
744 source_code = '''
745 int is_odd(int x) {
746 return (x%2);
747 }
749 int dec_to_hex(char hchar) {
750 if (hchar >= '0' && hchar <= '9') {
751 return (ord(hchar)-ord('0'));
752 } else {
753 return (10+ord(hchar)-ord('A'));
754 }
755 }
756 '''
757 for func in c_function.search_string(source_code):
758 print(f"{func.name} ({func.type}) args: {func.args}")
761 prints:
763 .. testoutput::
765 is_odd (int) args: [['int', 'x']]
766 dec_to_hex (int) args: [['char', 'hchar']]
767 """
768 return _NestedExpr(
769 opener, closer, content=content, ignore_expr=ignore_expr, **kwargs
770 )
773def _makeTags(tagStr, xml, suppress_LT=Suppress("<"), suppress_GT=Suppress(">")):
774 """Internal helper to construct opening and closing tag expressions,
775 given a tag name"""
776 if isinstance(tagStr, str_type):
777 resname = tagStr
778 tagStr = Keyword(tagStr, caseless=not xml)
779 else:
780 resname = tagStr.name
782 tagAttrName = Word(alphas, alphanums + "_-:")
783 if xml:
784 tagAttrValue = dbl_quoted_string.copy().set_parse_action(remove_quotes)
785 openTag = (
786 suppress_LT
787 + tagStr("tag")
788 + Dict(ZeroOrMore(Group(tagAttrName + Suppress("=") + tagAttrValue)))
789 + Opt("/", default=[False])("empty").set_parse_action(
790 lambda s, l, t: t[0] == "/"
791 )
792 + suppress_GT
793 )
794 else:
795 tagAttrValue = quoted_string.copy().set_parse_action(remove_quotes) | Word(
796 printables, exclude_chars=">"
797 )
798 openTag = (
799 suppress_LT
800 + tagStr("tag")
801 + Dict(
802 ZeroOrMore(
803 Group(
804 tagAttrName.set_parse_action(lambda t: t[0].lower())
805 + Opt(Suppress("=") + tagAttrValue)
806 )
807 )
808 )
809 + Opt("/", default=[False])("empty").set_parse_action(
810 lambda s, l, t: t[0] == "/"
811 )
812 + suppress_GT
813 )
814 closeTag = Combine(Literal("</") + tagStr + ">", adjacent=False)
816 openTag.set_name(f"<{resname}>")
817 # add start<tagname> results name in parse action now that ungrouped names are not reported at two levels
818 openTag.add_parse_action(
819 lambda t: t.__setitem__(
820 "start" + "".join(resname.replace(":", " ").title().split()), t.copy()
821 )
822 )
823 closeTag = closeTag(
824 "end" + "".join(resname.replace(":", " ").title().split())
825 ).set_name(f"</{resname}>")
826 openTag.tag = resname
827 closeTag.tag = resname
828 openTag.tag_body = SkipTo(closeTag())
829 return openTag, closeTag
832def make_html_tags(
833 tag_str: Union[str, ParserElement],
834) -> tuple[ParserElement, ParserElement]:
835 """Helper to construct opening and closing tag expressions for HTML,
836 given a tag name. Matches tags in either upper or lower case,
837 attributes with namespaces and with quoted or unquoted values.
839 Example:
841 .. testcode::
843 text = '<td>More info at the <a href="https://github.com/pyparsing/pyparsing/wiki">pyparsing</a> wiki page</td>'
844 # make_html_tags returns pyparsing expressions for the opening and
845 # closing tags as a 2-tuple
846 a, a_end = make_html_tags("A")
847 link_expr = a + SkipTo(a_end)("link_text") + a_end
849 for link in link_expr.search_string(text):
850 # attributes in the <A> tag (like "href" shown here) are
851 # also accessible as named results
852 print(link.link_text, '->', link.href)
854 prints:
856 .. testoutput::
858 pyparsing -> https://github.com/pyparsing/pyparsing/wiki
859 """
860 return _makeTags(tag_str, False)
863def make_xml_tags(
864 tag_str: Union[str, ParserElement],
865) -> tuple[ParserElement, ParserElement]:
866 """Helper to construct opening and closing tag expressions for XML,
867 given a tag name. Matches tags only in the given upper/lower case.
869 Example: similar to :class:`make_html_tags`
870 """
871 return _makeTags(tag_str, True)
874any_open_tag: ParserElement
875any_close_tag: ParserElement
876any_open_tag, any_close_tag = make_html_tags(
877 Word(alphas, alphanums + "_:").set_name("any tag")
878)
880_htmlEntityMap = {k.rstrip(";"): v for k, v in html.entities.html5.items()}
881_most_common_entities = "nbsp lt gt amp quot apos cent pound euro copy".replace(
882 " ", "|"
883)
884common_html_entity = Regex(
885 lambda: f"&(?P<entity>{_most_common_entities}|{make_compressed_re(_htmlEntityMap)});"
886).set_name("common HTML entity")
889def replace_html_entity(s, l, t):
890 """Helper parser action to replace common HTML entities with their special characters"""
891 return _htmlEntityMap.get(t.entity)
894class OpAssoc(Enum):
895 """Enumeration of operator associativity
896 - used in constructing InfixNotationOperatorSpec for :class:`infix_notation`"""
898 LEFT = 1
899 RIGHT = 2
902InfixNotationOperatorArgType = Union[
903 ParserElement, str, tuple[Union[ParserElement, str], Union[ParserElement, str]]
904]
905InfixNotationOperatorSpec = Union[
906 tuple[
907 InfixNotationOperatorArgType,
908 int,
909 OpAssoc,
910 typing.Optional[ParseAction],
911 ],
912 tuple[
913 InfixNotationOperatorArgType,
914 int,
915 OpAssoc,
916 ],
917]
920class _ParserState(Enum):
921 EXPECT_OPERAND = auto()
922 EXPECT_OPERATOR = auto()
925class _OpType(Enum):
926 PREFIX = auto()
927 POSTFIX = auto()
928 INFIX2 = auto()
929 INFIX2_RIGHT = auto()
930 TERNARY_LEFT_STAGE1 = auto()
931 TERNARY_LEFT_STAGE2 = auto()
932 TERNARY_RIGHT_STAGE1 = auto()
933 TERNARY_RIGHT_STAGE2 = auto()
934 LPAR = auto()
937class _InfixNotationOperatorSpec(NamedTuple):
938 op: InfixNotationOperatorArgType
939 arity: int
940 op_type: _OpType
941 parse_action: typing.Optional[list[Callable]]
944class _InfixNotation(ParseElementEnhance):
945 """
946 An iterative / stack-based implementation of infix_notation (using operator-precedence /
947 shunting-yard style evaluation with an explicit stack).
948 Prevents Python recursion limits from being exceeded on deeply nested expressions or long chains.
949 """
950 def __init__(
951 self,
952 base_expr: ParserElement,
953 op_list: list[_InfixNotationOperatorSpec],
954 lpar: Union[str, ParserElement] = Suppress("("),
955 rpar: Union[str, ParserElement] = Suppress(")"),
956 ):
957 from .core import _trim_arity
959 while isinstance(base_expr, _InfixNotation):
960 op_list[:0] = base_expr.op_list[:] # type: ignore[has-type]
961 base_expr = base_expr.base_expr # type: ignore[has-type]
963 super().__init__(base_expr, savelist=True)
964 self.base_expr = base_expr
965 self.op_list = op_list
966 self.lpar = Suppress(lpar) if isinstance(lpar, str) else lpar
967 self.rpar = Suppress(rpar) if isinstance(rpar, str) else rpar
968 self.keep_parens = not (isinstance(self.lpar, Suppress) and isinstance(self.rpar, Suppress))
969 self.errmsg = None
971 # Classify operators by arity and associativity
972 # Precedence is defined by order in op_list (higher index in op_list = lower precedence)
973 # We assign higher numeric precedence to earlier entries in op_list
974 self.prefix_ops: list[_InfixNotationOperatorSpec] = [] # (expr, prec, assoc, pa_list)
975 self.postfix_ops: list[_InfixNotationOperatorSpec] = [] # (expr, prec, assoc, pa_list)
976 self.infix_ops = [] # type: ignore[var-annotated]
978 total_ops = len(op_list)
979 for idx, oper_def in enumerate(op_list):
980 op_expr, arity, assoc, *pa_opt = (oper_def + (None,))[:4]
981 pa = pa_opt[0] if pa_opt else None
982 pa_list = (
983 [_trim_arity(f) for f in pa]
984 if isinstance(pa, (tuple, list))
985 else ([_trim_arity(pa)] if pa is not None else [])
986 )
988 if not 1 <= arity <= 3:
989 raise ValueError("operator must be unary (1), binary (2), or ternary (3)")
991 if assoc not in (OpAssoc.LEFT, OpAssoc.RIGHT):
992 raise ValueError("operator must indicate right or left associativity")
994 # Precedence: top of list has highest precedence
995 prec = (total_ops - idx) * 10
997 if arity == 1:
998 if isinstance(op_expr, str_type):
999 op_expr = Literal(op_expr)
1000 if assoc is OpAssoc.RIGHT:
1001 self.prefix_ops.append((op_expr, prec, assoc, pa_list))
1002 else:
1003 self.postfix_ops.append((op_expr, prec, assoc, pa_list))
1004 elif arity == 2:
1005 if op_expr is not None and isinstance(op_expr, str_type):
1006 op_expr = Literal(op_expr)
1007 self.infix_ops.append((op_expr, prec, assoc, pa_list, 2, None))
1008 elif arity == 3:
1009 if not isinstance(op_expr, (tuple, list)) or len(op_expr) != 2:
1010 raise ValueError(
1011 "if numterms=3, opExpr must be a tuple or list of two expressions"
1012 )
1013 op1, op2 = op_expr
1014 if isinstance(op1, str_type):
1015 op1 = Literal(op1)
1016 if isinstance(op2, str_type):
1017 op2 = Literal(op2)
1018 self.infix_ops.append((op1, prec, assoc, pa_list, 3, op2))
1020 def _generateDefaultName(self) -> str:
1021 return f"{self.base_expr} infix expression"
1023 def _run_parse_actions(self, pa_list, instring, loc, tokens):
1024 ret_tokens = tokens
1025 for fn in pa_list:
1026 res = fn(instring, loc, ret_tokens)
1027 if res is not None and res is not ret_tokens:
1028 if isinstance(res, (ParseResults, list, tuple)):
1029 ret_tokens = ParseResults(res, aslist=True)
1030 else:
1031 return res
1032 return ret_tokens
1034 def _apply_operator(self, op_info, operand_stack, instring):
1035 op_type = op_info["type"]
1036 pa_list = op_info.get("pa", [])
1037 start_loc = op_info.get("loc", 0)
1039 if op_type is _OpType.PREFIX:
1040 op_tok = op_info["op"]
1041 arg = operand_stack.pop()
1042 tokens = []
1043 if isinstance(op_tok, (list, ParseResults)):
1044 tokens.extend(op_tok)
1045 elif op_tok is not None:
1046 tokens.append(op_tok)
1047 tokens.append(arg)
1048 tokens_pr = ParseResults(tokens)
1049 if pa_list:
1050 res = self._run_parse_actions(pa_list, instring, start_loc, ParseResults([tokens_pr]))
1051 else:
1052 res = tokens_pr
1053 operand_stack.append(res)
1055 elif op_type is _OpType.POSTFIX:
1056 arg = operand_stack.pop()
1057 if isinstance(arg, ParseResults):
1058 res_pr = arg.copy()
1059 elif isinstance(arg, list):
1060 res_pr = ParseResults(arg)
1061 else:
1062 res_pr = ParseResults([arg])
1063 for op_tok in op_info["ops"]:
1064 if isinstance(op_tok, ParseResults):
1065 res_pr += op_tok
1066 elif isinstance(op_tok, list):
1067 res_pr.extend(op_tok)
1068 elif op_tok is not None:
1069 res_pr.append(op_tok)
1070 if pa_list:
1071 res = self._run_parse_actions(pa_list, instring, start_loc, ParseResults([res_pr]))
1072 else:
1073 res = res_pr
1074 operand_stack.append(res)
1076 elif op_type is _OpType.INFIX2:
1077 ops = op_info["ops"]
1078 num_ops = len(ops)
1079 args = [operand_stack.pop() for _ in range(num_ops + 1)]
1080 args.reverse()
1081 tokens = [args[0]]
1082 for i in range(num_ops):
1083 op_tok = ops[i]
1084 if isinstance(op_tok, (list, ParseResults)):
1085 tokens.extend(op_tok)
1086 elif op_tok is not None:
1087 tokens.append(op_tok)
1088 tokens.append(args[i + 1])
1089 tokens_pr = ParseResults(tokens)
1090 if pa_list:
1091 res = self._run_parse_actions(pa_list, instring, start_loc, ParseResults([tokens_pr]))
1092 else:
1093 res = tokens_pr
1094 operand_stack.append(res)
1096 elif op_type is _OpType.INFIX2_RIGHT:
1097 op_tok = op_info["op"]
1098 right = operand_stack.pop()
1099 left = operand_stack.pop()
1100 tokens = [left]
1101 if isinstance(op_tok, (list, ParseResults)):
1102 tokens.extend(op_tok)
1103 elif op_tok is not None:
1104 tokens.append(op_tok)
1105 tokens.append(right)
1106 tokens_pr = ParseResults(tokens)
1107 if pa_list:
1108 res = self._run_parse_actions(pa_list, instring, start_loc, ParseResults([tokens_pr]))
1109 else:
1110 res = tokens_pr
1111 operand_stack.append(res)
1113 elif op_type is _OpType.TERNARY_LEFT_STAGE2:
1114 ops_list = op_info["ops"]
1115 num_ops = len(ops_list)
1116 num_operands = 2 * num_ops + 1
1117 args = [operand_stack.pop() for _ in range(num_operands)]
1118 args.reverse()
1119 tokens = [args[0]]
1120 for i in range(num_ops):
1121 op1_tok, op2_tok = ops_list[i]
1122 if isinstance(op1_tok, (list, ParseResults)):
1123 tokens.extend(op1_tok)
1124 elif op1_tok is not None:
1125 tokens.append(op1_tok)
1126 tokens.append(args[2 * i + 1])
1127 if isinstance(op2_tok, (list, ParseResults)):
1128 tokens.extend(op2_tok)
1129 elif op2_tok is not None:
1130 tokens.append(op2_tok)
1131 tokens.append(args[2 * i + 2])
1132 tokens_pr = ParseResults(tokens)
1133 if pa_list:
1134 res = self._run_parse_actions(pa_list, instring, start_loc, ParseResults([tokens_pr]))
1135 else:
1136 res = tokens_pr
1137 operand_stack.append(res)
1139 elif op_type is _OpType.TERNARY_RIGHT_STAGE2:
1140 op1_tok = op_info["op1"]
1141 op2_tok = op_info["op2"]
1142 arg3 = operand_stack.pop()
1143 arg2 = operand_stack.pop()
1144 arg1 = operand_stack.pop()
1145 tokens = [arg1]
1146 if isinstance(op1_tok, (list, ParseResults)):
1147 tokens.extend(op1_tok)
1148 elif op1_tok is not None:
1149 tokens.append(op1_tok)
1150 tokens.append(arg2)
1151 if isinstance(op2_tok, (list, ParseResults)):
1152 tokens.extend(op2_tok)
1153 elif op2_tok is not None:
1154 tokens.append(op2_tok)
1155 tokens.append(arg3)
1156 tokens_pr = ParseResults(tokens)
1157 if pa_list:
1158 res = self._run_parse_actions(pa_list, instring, start_loc, ParseResults([tokens_pr]))
1159 else:
1160 res = tokens_pr
1161 operand_stack.append(res)
1163 return bool(pa_list)
1165 def parseImpl(self, instring, loc, do_actions=True):
1166 operand_stack = []
1167 operator_stack = []
1168 last_pa_applied = [False]
1170 def reduce_operators(min_prec):
1171 while operator_stack:
1172 top = operator_stack[-1]
1173 if top["type"] in (_OpType.LPAR, _OpType.TERNARY_LEFT_STAGE1, _OpType.TERNARY_RIGHT_STAGE1):
1174 break
1175 if top["prec"] > min_prec:
1176 op_info = operator_stack.pop()
1177 last_pa_applied[0] = self._apply_operator(op_info, operand_stack, instring)
1178 else:
1179 break
1181 state = _ParserState.EXPECT_OPERAND
1182 paren_depth = 0
1184 while True:
1185 try:
1186 loc = self.preParse(instring, loc)
1187 except ParseException:
1188 pass
1190 if state is _ParserState.EXPECT_OPERAND:
1191 # 1. Try prefix operators
1192 matched_prefix = False
1193 for op_expr, prec, assoc, pa_list in self.prefix_ops:
1194 try:
1195 next_loc, toks = op_expr._parse(instring, loc, do_actions=do_actions)
1196 op_tok = toks.as_list()[0] if len(toks) == 1 else toks.as_list()
1197 operator_stack.append({
1198 "type": _OpType.PREFIX,
1199 "op": op_tok,
1200 "prec": prec,
1201 "assoc": assoc,
1202 "pa": pa_list if do_actions else [],
1203 "loc": loc,
1204 })
1205 loc = next_loc
1206 matched_prefix = True
1207 break
1208 except ParseException:
1209 pass
1210 if matched_prefix:
1211 continue
1213 # 2. Try base_expr
1214 try:
1215 next_loc, base_toks = self.base_expr._parse(instring, loc, do_actions=do_actions)
1216 if not base_toks and not base_toks._tokdict:
1217 loc = next_loc
1218 continue
1219 operand = base_toks[0] if len(base_toks) == 1 else base_toks
1220 operand_stack.append(operand)
1221 loc = next_loc
1222 state = _ParserState.EXPECT_OPERATOR
1223 continue
1224 except ParseException:
1225 pass
1227 # 3. Try lpar
1228 try:
1229 next_loc, lpar_toks = self.lpar._parse(instring, loc, do_actions=do_actions)
1230 operator_stack.append({
1231 "type": _OpType.LPAR,
1232 "paren_toks": lpar_toks.as_list(),
1233 "loc": loc,
1234 })
1235 paren_depth += 1
1236 loc = next_loc
1237 continue
1238 except ParseException:
1239 raise ParseException(instring, loc, f"Expected {self.base_expr}")
1241 elif state is _ParserState.EXPECT_OPERATOR:
1242 # 1. Try postfix operators
1243 matched_postfix = False
1244 for op_expr, prec, assoc, pa_list in self.postfix_ops:
1245 try:
1246 next_loc, toks = op_expr._parse(instring, loc, do_actions=do_actions)
1247 op_tok = toks
1248 if (
1249 operator_stack
1250 and operator_stack[-1]["type"] is _OpType.POSTFIX
1251 and operator_stack[-1]["prec"] == prec
1252 and operator_stack[-1]["assoc"] is OpAssoc.LEFT
1253 ):
1254 operator_stack[-1]["ops"].append(op_tok)
1255 else:
1256 reduce_operators(prec)
1257 operator_stack.append({
1258 "type": _OpType.POSTFIX,
1259 "ops": [op_tok],
1260 "prec": prec,
1261 "assoc": assoc,
1262 "pa": pa_list if do_actions else [],
1263 "loc": loc,
1264 })
1265 loc = next_loc
1266 matched_postfix = True
1267 break
1268 except ParseException:
1269 pass
1270 if matched_postfix:
1271 continue
1273 # 2. Try rpar if inside parens
1274 if paren_depth > 0:
1275 try:
1276 next_loc, rpar_toks = self.rpar._parse(instring, loc, do_actions=do_actions)
1277 reduce_operators(-1) # reduce all operators inside this paren
1278 if operator_stack and operator_stack[-1]["type"] is _OpType.LPAR:
1279 lpar_info = operator_stack.pop()
1280 if self.keep_parens:
1281 top_val = operand_stack.pop()
1282 operand_stack.append(ParseResults([*lpar_info["paren_toks"], top_val, *rpar_toks.as_list()]))
1283 paren_depth -= 1
1284 loc = next_loc
1285 continue
1286 except ParseException:
1287 pass
1289 # 3. Try matching op2 for an open ternary operator (STAGE1)
1290 matched_op2 = False
1291 for i in range(len(operator_stack) - 1, -1, -1):
1292 item = operator_stack[i]
1293 if item["type"] is _OpType.LPAR:
1294 break
1295 if item["type"] in (_OpType.TERNARY_LEFT_STAGE1, _OpType.TERNARY_RIGHT_STAGE1):
1296 try:
1297 next_loc, toks = item["op2_expr"]._parse(instring, loc, do_actions=do_actions)
1298 op2_tok = toks.as_list()[0] if len(toks) == 1 else toks.as_list()
1299 while operator_stack and operator_stack[-1] is not item:
1300 op_info = operator_stack.pop()
1301 self._apply_operator(op_info, operand_stack, instring)
1302 if item["type"] is _OpType.TERNARY_LEFT_STAGE1:
1303 item["type"] = _OpType.TERNARY_LEFT_STAGE2
1304 item["ops"][-1].append(op2_tok)
1305 else:
1306 item["type"] = _OpType.TERNARY_RIGHT_STAGE2
1307 item["op2"] = op2_tok
1308 loc = next_loc
1309 matched_op2 = True
1310 state = _ParserState.EXPECT_OPERAND
1311 break
1312 except ParseException:
1313 pass
1314 if matched_op2:
1315 continue
1317 def _check_operand_follows(check_loc):
1318 for p_op, _, _, _ in self.prefix_ops:
1319 try:
1320 p_op._parse(instring, check_loc, do_actions=False)
1321 return True
1322 except ParseException:
1323 pass
1324 try:
1325 self.base_expr._parse(instring, check_loc, do_actions=False)
1326 return True
1327 except ParseException:
1328 pass
1329 try:
1330 self.lpar._parse(instring, check_loc, do_actions=False)
1331 return True
1332 except ParseException:
1333 pass
1334 return False
1336 # 4. Try infix binary and ternary operators
1337 matched_infix = False
1338 for op_expr, prec, assoc, pa_list, arity, op2_expr in self.infix_ops:
1339 if arity == 2:
1340 if op_expr is None:
1341 if not _check_operand_follows(loc):
1342 continue
1343 next_loc = loc
1344 op_tok = None
1345 else:
1346 try:
1347 next_loc, toks = op_expr._parse(instring, loc, do_actions=do_actions)
1348 except ParseException:
1349 continue
1350 if next_loc == loc:
1351 if not _check_operand_follows(loc):
1352 continue
1353 op_tok = (toks.as_list()[0] if len(toks) == 1 else toks.as_list()) if toks else None
1354 else:
1355 op_tok = toks.as_list()[0] if len(toks) == 1 else toks.as_list()
1357 if assoc is OpAssoc.LEFT:
1358 reduce_operators(prec)
1359 if (
1360 operator_stack
1361 and operator_stack[-1]["type"] is _OpType.INFIX2
1362 and operator_stack[-1]["prec"] == prec
1363 ):
1364 operator_stack[-1]["ops"].append(op_tok)
1365 else:
1366 operator_stack.append({
1367 "type": _OpType.INFIX2,
1368 "ops": [op_tok],
1369 "prec": prec,
1370 "assoc": assoc,
1371 "pa": pa_list if do_actions else [],
1372 "loc": loc,
1373 })
1374 else:
1375 reduce_operators(prec)
1376 operator_stack.append({
1377 "type": _OpType.INFIX2_RIGHT,
1378 "op": op_tok,
1379 "prec": prec,
1380 "assoc": assoc,
1381 "pa": pa_list if do_actions else [],
1382 "loc": loc,
1383 })
1384 loc = next_loc
1385 matched_infix = True
1386 state = _ParserState.EXPECT_OPERAND
1387 break
1388 elif arity == 3:
1389 try:
1390 next_loc, toks = op_expr._parse(instring, loc, do_actions=do_actions)
1391 if next_loc == loc and not _check_operand_follows(loc):
1392 continue
1393 op_tok = toks.as_list()[0] if len(toks) == 1 else toks.as_list()
1394 if assoc is OpAssoc.LEFT:
1395 if (
1396 operator_stack
1397 and operator_stack[-1]["type"] is _OpType.TERNARY_LEFT_STAGE2
1398 and operator_stack[-1]["prec"] == prec
1399 ):
1400 operator_stack[-1]["type"] = _OpType.TERNARY_LEFT_STAGE1
1401 operator_stack[-1]["ops"].append([op_tok])
1402 else:
1403 reduce_operators(prec)
1404 operator_stack.append({
1405 "type": _OpType.TERNARY_LEFT_STAGE1,
1406 "ops": [[op_tok]],
1407 "op2_expr": op2_expr,
1408 "prec": prec,
1409 "assoc": assoc,
1410 "pa": pa_list if do_actions else [],
1411 "loc": loc,
1412 })
1413 else:
1414 reduce_operators(prec)
1415 operator_stack.append({
1416 "type": _OpType.TERNARY_RIGHT_STAGE1,
1417 "op1": op_tok,
1418 "op2_expr": op2_expr,
1419 "prec": prec,
1420 "assoc": assoc,
1421 "pa": pa_list if do_actions else [],
1422 "loc": loc,
1423 })
1424 loc = next_loc
1425 matched_infix = True
1426 state = _ParserState.EXPECT_OPERAND
1427 break
1428 except ParseException:
1429 pass
1430 if matched_infix:
1431 continue
1433 # No more operators can be consumed at this level
1434 break
1436 # Reduce remaining operators
1437 reduce_operators(-1)
1439 if paren_depth != 0 or len(operand_stack) != 1 or operator_stack:
1440 raise ParseException(instring, loc, "Unbalanced parentheses or expression syntax error")
1442 final_result = operand_stack.pop()
1443 if last_pa_applied[0] and isinstance(final_result, ParseResults):
1444 return loc, final_result
1445 else:
1446 return loc, ParseResults([final_result])
1449def infix_notation(base_expr, op_list, lpar="(", rpar=")"):
1450 """Helper method for constructing grammars of expressions made up of
1451 operators working in a precedence hierarchy. Operators may be unary
1452 or binary, left- or right-associative. Parse actions can also be
1453 attached to operator expressions. The generated parser will also
1454 recognize the use of parentheses to override operator precedences
1455 (see example below).
1457 Note: if you define a deep operator list, you may see performance
1458 issues when using infix_notation. See
1459 :class:`ParserElement.enable_packrat` for a mechanism to potentially
1460 improve your parser performance.
1462 Parameters:
1464 :param base_expr: expression representing the most basic operand to
1465 be used in the expression
1466 :param op_list: list of tuples, one for each operator precedence level
1467 in the expression grammar; each tuple is of the form ``(op_expr,
1468 num_operands, right_left_assoc, (optional)parse_action)``, where:
1470 - ``op_expr`` is the pyparsing expression for the operator; may also
1471 be a string, which will be converted to a Literal; if ``num_operands``
1472 is 3, ``op_expr`` is a tuple of two expressions, for the two
1473 operators separating the 3 terms
1474 - ``num_operands`` is the number of terms for this operator (must be 1,
1475 2, or 3)
1476 - ``right_left_assoc`` is the indicator whether the operator is right
1477 or left associative, using the pyparsing-defined constants
1478 ``OpAssoc.RIGHT`` and ``OpAssoc.LEFT``.
1479 - ``parse_action`` is the parse action to be associated with
1480 expressions matching this operator expression (the parse action
1481 tuple member may be omitted); if the parse action is passed
1482 a tuple or list of functions, this is equivalent to calling
1483 ``set_parse_action(*fn)``
1484 (:class:`ParserElement.set_parse_action`)
1486 :param lpar: expression for matching left-parentheses; if passed as a
1487 str, then will be parsed as ``Suppress(lpar)``. If lpar is passed as
1488 an expression (such as ``Literal('(')``), then it will be kept in
1489 the parsed results, and grouped with them. (default= ``Suppress('(')``)
1490 :param rpar: expression for matching right-parentheses; if passed as a
1491 str, then will be parsed as ``Suppress(rpar)``. If rpar is passed as
1492 an expression (such as ``Literal(')')``), then it will be kept in
1493 the parsed results, and grouped with them. (default= ``Suppress(')')``)
1495 Example:
1497 .. testcode::
1499 # simple example of four-function arithmetic with ints and
1500 # variable names
1501 integer = pyparsing_common.signed_integer
1502 varname = pyparsing_common.identifier
1504 arith_expr = infix_notation(integer | varname,
1505 [
1506 ('-', 1, OpAssoc.RIGHT),
1507 (one_of('* /'), 2, OpAssoc.LEFT),
1508 (one_of('+ -'), 2, OpAssoc.LEFT),
1509 ])
1511 arith_expr.run_tests('''
1512 5+3*6
1513 (5+3)*6
1514 (5+x)*y
1515 -2--11
1516 ''', full_dump=False)
1518 prints:
1520 .. testoutput::
1521 :options: +NORMALIZE_WHITESPACE
1524 5+3*6
1525 [[5, '+', [3, '*', 6]]]
1527 (5+3)*6
1528 [[[5, '+', 3], '*', 6]]
1530 (5+x)*y
1531 [[[5, '+', 'x'], '*', 'y']]
1533 -2--11
1534 [[['-', 2], '-', ['-', 11]]]
1535 """
1536 return _InfixNotation(base_expr, op_list, lpar=lpar, rpar=rpar)
1539def indentedBlock(blockStatementExpr, indentStack, indent=True, backup_stacks=[]):
1540 """
1541 .. deprecated:: 3.0.0
1542 Use the :class:`IndentedBlock` class instead. Note that `IndentedBlock`
1543 has a difference method signature.
1545 Helper method for defining space-delimited indentation blocks,
1546 such as those used to define block statements in Python source code.
1548 :param blockStatementExpr: expression defining syntax of statement that
1549 is repeated within the indented block
1551 :param indentStack: list created by caller to manage indentation stack
1552 (multiple ``statementWithIndentedBlock`` expressions within a single
1553 grammar should share a common ``indentStack``)
1555 :param indent: boolean indicating whether block must be indented beyond
1556 the current level; set to ``False`` for block of left-most statements
1558 A valid block must contain at least one ``blockStatement``.
1560 (Note that indentedBlock uses internal parse actions which make it
1561 incompatible with packrat parsing.)
1563 Example:
1565 .. testcode::
1567 data = '''
1568 def A(z):
1569 A1
1570 B = 100
1571 G = A2
1572 A2
1573 A3
1574 B
1575 def BB(a,b,c):
1576 BB1
1577 def BBA():
1578 bba1
1579 bba2
1580 bba3
1581 C
1582 D
1583 def spam(x,y):
1584 def eggs(z):
1585 pass
1586 '''
1588 indentStack = [1]
1589 stmt = Forward()
1591 identifier = Word(alphas, alphanums)
1592 funcDecl = ("def" + identifier + Group("(" + Opt(delimitedList(identifier)) + ")") + ":")
1593 func_body = indentedBlock(stmt, indentStack)
1594 funcDef = Group(funcDecl + func_body)
1596 rvalue = Forward()
1597 funcCall = Group(identifier + "(" + Opt(delimitedList(rvalue)) + ")")
1598 rvalue << (funcCall | identifier | Word(nums))
1599 assignment = Group(identifier + "=" + rvalue)
1600 stmt << (funcDef | assignment | identifier)
1602 module_body = stmt[1, ...]
1604 parseTree = module_body.parseString(data)
1605 parseTree.pprint()
1607 prints:
1609 .. testoutput::
1611 [['def',
1612 'A',
1613 ['(', 'z', ')'],
1614 ':',
1615 [['A1'], [['B', '=', '100']], [['G', '=', 'A2']], ['A2'], ['A3']]],
1616 'B',
1617 ['def',
1618 'BB',
1619 ['(', 'a', 'b', 'c', ')'],
1620 ':',
1621 [['BB1'], [['def', 'BBA', ['(', ')'], ':', [['bba1'], ['bba2'], ['bba3']]]]]],
1622 'C',
1623 'D',
1624 ['def',
1625 'spam',
1626 ['(', 'x', 'y', ')'],
1627 ':',
1628 [[['def', 'eggs', ['(', 'z', ')'], ':', [['pass']]]]]]]
1629 """
1630 warnings.warn(
1631 f"{'indentedBlock'!r} deprecated - use {'IndentedBlock'!r}",
1632 PyparsingDeprecationWarning,
1633 stacklevel=2,
1634 )
1636 backup_stacks.append(indentStack[:])
1638 def reset_stack():
1639 indentStack[:] = backup_stacks[-1]
1641 def checkPeerIndent(s, l, t):
1642 if l >= len(s):
1643 return
1644 curCol = col(l, s)
1645 if curCol != indentStack[-1]:
1646 if curCol > indentStack[-1]:
1647 raise ParseException(s, l, "illegal nesting")
1648 raise ParseException(s, l, "not a peer entry")
1650 def checkSubIndent(s, l, t):
1651 curCol = col(l, s)
1652 if curCol > indentStack[-1]:
1653 indentStack.append(curCol)
1654 else:
1655 raise ParseException(s, l, "not a subentry")
1657 def checkUnindent(s, l, t):
1658 if l >= len(s):
1659 return
1660 curCol = col(l, s)
1661 if not (indentStack and curCol in indentStack):
1662 raise ParseException(s, l, "not an unindent")
1663 if curCol < indentStack[-1]:
1664 indentStack.pop()
1666 NL = OneOrMore(LineEnd().set_whitespace_chars("\t ").suppress())
1667 INDENT = (Empty() + Empty().set_parse_action(checkSubIndent)).set_name("INDENT")
1668 PEER = Empty().set_parse_action(checkPeerIndent).set_name("")
1669 UNDENT = Empty().set_parse_action(checkUnindent).set_name("UNINDENT")
1670 if indent:
1671 smExpr = Group(
1672 Opt(NL)
1673 + INDENT
1674 + OneOrMore(PEER + Group(blockStatementExpr) + Opt(NL))
1675 + UNDENT
1676 )
1677 else:
1678 smExpr = Group(
1679 Opt(NL)
1680 + OneOrMore(PEER + Group(blockStatementExpr) + Opt(NL))
1681 + Opt(UNDENT)
1682 )
1684 # add a parse action to remove backup_stack from list of backups
1685 smExpr.add_parse_action(
1686 lambda: backup_stacks.pop(-1) and None if backup_stacks else None
1687 )
1688 smExpr.set_fail_action(lambda a, b, c, d: reset_stack())
1689 blockStatementExpr.ignore(_bslash + LineEnd())
1690 return smExpr.set_name("indented block")
1693# it's easy to get these comment structures wrong - they're very common,
1694# so may as well make them available
1695c_style_comment = Regex(r"/\*(?:[^*]|\*(?!/))*\*\/").set_name("C style comment")
1696"Comment of the form ``/* ... */``"
1698html_comment = Regex(r"<!--[\s\S]*?-->").set_name("HTML comment")
1699"Comment of the form ``<!-- ... -->``"
1701rest_of_line = Regex(r".*").leave_whitespace().set_name("rest of line")
1702dbl_slash_comment = Regex(r"//(?:\\\n|[^\n])*").set_name("// comment")
1703"Comment of the form ``// ... (to end of line)``"
1705cpp_style_comment = Regex(
1706 r"(?:/\*(?:[^*]|\*(?!/))*\*\/)|(?://(?:\\\n|[^\n])*)"
1707).set_name("C++ style comment")
1708"Comment of either form :class:`c_style_comment` or :class:`dbl_slash_comment`"
1710java_style_comment = cpp_style_comment
1711"Same as :class:`cpp_style_comment`"
1713python_style_comment = Regex(r"#.*").set_name("Python style comment")
1714"Comment of the form ``# ... (to end of line)``"
1717# build list of built-in expressions, for future reference if a global default value
1718# gets updated
1719_builtin_exprs: list[ParserElement] = [
1720 v for v in vars().values() if isinstance(v, ParserElement)
1721]
1724# compatibility function, superseded by DelimitedList class
1725def delimited_list(
1726 expr: Union[str, ParserElement],
1727 delim: Union[str, ParserElement] = ",",
1728 combine: bool = False,
1729 min: typing.Optional[int] = None,
1730 max: typing.Optional[int] = None,
1731 *,
1732 allow_trailing_delim: bool = False,
1733) -> ParserElement:
1734 """
1735 .. deprecated:: 3.1.0
1736 Use the :class:`DelimitedList` class instead.
1737 """
1738 return DelimitedList(
1739 expr, delim, combine, min, max, allow_trailing_delim=allow_trailing_delim
1740 )
1743# Compatibility synonyms
1744# fmt: off
1745opAssoc = OpAssoc
1746anyOpenTag = any_open_tag
1747anyCloseTag = any_close_tag
1748commonHTMLEntity = common_html_entity
1749cStyleComment = c_style_comment
1750htmlComment = html_comment
1751restOfLine = rest_of_line
1752dblSlashComment = dbl_slash_comment
1753cppStyleComment = cpp_style_comment
1754javaStyleComment = java_style_comment
1755pythonStyleComment = python_style_comment
1756delimitedList = replaced_by_pep8("delimitedList", DelimitedList)
1757delimited_list = replaced_by_pep8("delimited_list", DelimitedList)
1758countedArray = replaced_by_pep8("countedArray", counted_array)
1759matchPreviousLiteral = replaced_by_pep8("matchPreviousLiteral", match_previous_literal)
1760matchPreviousExpr = replaced_by_pep8("matchPreviousExpr", match_previous_expr)
1761oneOf = replaced_by_pep8("oneOf", one_of)
1762dictOf = replaced_by_pep8("dictOf", dict_of)
1763originalTextFor = replaced_by_pep8("originalTextFor", original_text_for)
1764nestedExpr = replaced_by_pep8("nestedExpr", nested_expr)
1765makeHTMLTags = replaced_by_pep8("makeHTMLTags", make_html_tags)
1766makeXMLTags = replaced_by_pep8("makeXMLTags", make_xml_tags)
1767replaceHTMLEntity = replaced_by_pep8("replaceHTMLEntity", replace_html_entity)
1768infixNotation = replaced_by_pep8("infixNotation", infix_notation)
1769# fmt: on