1"""Token-related utilities"""
2
3# Copyright (c) IPython Development Team.
4# Distributed under the terms of the Modified BSD License.
5from __future__ import annotations
6
7import itertools
8import tokenize
9from io import StringIO
10from keyword import iskeyword
11from tokenize import TokenInfo
12from typing import NamedTuple
13from collections.abc import Callable
14from collections.abc import Generator
15
16
17class Token(NamedTuple):
18 token: int
19 text: str
20 start: int
21 end: int
22 line: str
23
24
25def generate_tokens(readline: Callable) -> Generator[TokenInfo]:
26 """wrap generate_tkens to catch EOF errors"""
27 try:
28 yield from tokenize.generate_tokens(readline)
29 except tokenize.TokenError:
30 # catch EOF error
31 return
32
33
34def generate_tokens_catch_errors(
35 readline, extra_errors_to_catch: list[str] | None = None
36):
37 default_errors_to_catch = [
38 "unterminated string literal",
39 "invalid non-printable character",
40 "after line continuation character",
41 # Since Python 3.12 the tokenizer raises TokenError for malformed
42 # number literals (e.g. 0b12, 0o1239, 1__2). Those are syntax
43 # errors, not incomplete input (see ipython/ipython#15320).
44 "invalid decimal literal",
45 "invalid binary literal",
46 "invalid octal literal",
47 "invalid hexadecimal literal",
48 "in binary literal",
49 "in octal literal",
50 ]
51 assert extra_errors_to_catch is None or isinstance(extra_errors_to_catch, list)
52 errors_to_catch = default_errors_to_catch + (extra_errors_to_catch or [])
53
54 tokens: list[TokenInfo] = []
55 try:
56 for token in tokenize.generate_tokens(readline):
57 tokens.append(token)
58 yield token
59 except tokenize.TokenError as exc:
60 if any(error in exc.args[0] for error in errors_to_catch):
61 if tokens:
62 start = tokens[-1].start[0], tokens[-1].end[0]
63 end = start
64 line = tokens[-1].line
65 else:
66 start = end = (1, 0)
67 line = ""
68 yield TokenInfo(tokenize.ERRORTOKEN, "", start, end, line)
69 else:
70 # Catch EOF
71 raise
72
73
74def line_at_cursor(cell: str, cursor_pos: int = 0) -> tuple[str, int]:
75 """Return the line in a cell at a given cursor position
76
77 Used for calling line-based APIs that don't support multi-line input, yet.
78
79 Parameters
80 ----------
81 cell : str
82 multiline block of text
83 cursor_pos : integer
84 the cursor position
85
86 Returns
87 -------
88 (line, offset): (string, integer)
89 The line with the current cursor, and the character offset of the start of the line.
90 """
91 offset = 0
92 lines = cell.splitlines(True)
93 for line in lines:
94 next_offset = offset + len(line)
95 if not line.endswith("\n"):
96 # If the last line doesn't have a trailing newline, treat it as if
97 # it does so that the cursor at the end of the line still counts
98 # as being on that line.
99 next_offset += 1
100 if next_offset > cursor_pos:
101 break
102 offset = next_offset
103 else:
104 line = ""
105 return line, offset
106
107
108def token_at_cursor(cell: str, cursor_pos: int = 0) -> str:
109 """Get the token at a given cursor
110
111 Used for introspection.
112
113 Function calls are prioritized, so the token for the callable will be returned
114 if the cursor is anywhere inside the call.
115
116 Parameters
117 ----------
118 cell : str
119 A block of Python code
120 cursor_pos : int
121 The location of the cursor in the block where the token should be found
122 """
123 names: list[str] = []
124 call_names: list[str] = []
125 closing_call_name: str | None = None
126 most_recent_outer_name: str | None = None
127
128 offsets = {1: 0} # lines start at 1
129 intersects_with_cursor = False
130 cur_token_is_name = False
131 tokens: list[Token | None] = [
132 Token(*tup) for tup in generate_tokens(StringIO(cell).readline)
133 ]
134 if not tokens:
135 return ""
136 for prev_tok, (tok, next_tok) in zip(
137 [None] + tokens, itertools.pairwise(tokens + [None])
138 ):
139 # token, text, start, end, line = tup
140 start_line, start_col = tok.start
141 end_line, end_col = tok.end
142 if end_line + 1 not in offsets:
143 # keep track of offsets for each line
144 lines = tok.line.splitlines(True)
145 for lineno, line in enumerate(lines, start_line + 1):
146 if lineno not in offsets:
147 offsets[lineno] = offsets[lineno - 1] + len(line)
148
149 closing_call_name = None
150
151 offset = offsets[start_line]
152 if offset + start_col > cursor_pos:
153 # current token starts after the cursor,
154 # don't consume it
155 break
156
157 if cur_token_is_name := tok.token == tokenize.NAME and not iskeyword(tok.text):
158 if (
159 names
160 and prev_tok
161 and prev_tok.token == tokenize.OP
162 and prev_tok.text == "."
163 ):
164 names[-1] = "{}.{}".format(names[-1], tok.text)
165 else:
166 names.append(tok.text)
167 if (
168 next_tok is not None
169 and next_tok.token == tokenize.OP
170 and next_tok.text == "="
171 ):
172 # don't inspect the lhs of an assignment
173 names.pop(-1)
174 cur_token_is_name = False
175 if not call_names:
176 most_recent_outer_name = names[-1] if names else None
177 elif tok.token == tokenize.OP:
178 if tok.text == "(" and names:
179 # if we are inside a function call, inspect the function
180 call_names.append(names[-1])
181 elif tok.text == ")" and call_names:
182 # keep track of the most recently popped call_name from the stack
183 closing_call_name = call_names.pop(-1)
184
185 if offsets[end_line] + end_col > cursor_pos:
186 # we found the cursor, stop reading
187 # if the current token intersects directly, use it instead of the call token
188 intersects_with_cursor = offsets[start_line] + start_col <= cursor_pos
189 break
190
191 if cur_token_is_name and intersects_with_cursor:
192 return names[-1]
193 # if the cursor isn't directly over a name token, use the most recent
194 # call name if we can find one
195 elif closing_call_name:
196 # if we're on a ")", use the most recently popped call name
197 return closing_call_name
198 elif call_names:
199 # otherwise, look for the most recent call name in the stack
200 return call_names[-1]
201 elif most_recent_outer_name:
202 # if we've popped all the call names, use the most recently-seen
203 # outer name
204 return most_recent_outer_name
205 elif names:
206 # failing that, use the most recently seen name
207 return names[-1]
208 else:
209 # give up
210 return ""