import ast
import keyword
from token_utils import py_tokenize
_token_format = "type={type} string={string} start={start} end={end} line={line}"
[docs]
class Token:
"""Token as generated from Python's tokenize.generate_tokens written here in
a more convenient form, and with some custom methods.
The various parameters are::
type: token type
string: the token written as a string
start = (start_row, start_col)
end = (end_row, end_col)
line: entire line of code where the token is found.
Token instances are mutable objects. Therefore, given a list of tokens,
we can change the value of any token's attribute, untokenize the list and
automatically obtain a transformed source.
"""
[docs]
def __init__(self, token):
"""Initializes using a token produced by Python's tokenize function as input."""
self.type = token[0]
self.string = token[1]
self.start = self.start_row, self.start_col = token[2]
self.end = self.end_row, self.end_col = token[3]
self.line = token[4]
[docs]
def __eq__(self, other):
"""Compares a Token with another object; returns true if
self.string == other.string or if self.string == other.
"""
if hasattr(other, "string"):
return self.string == other.string
elif isinstance(other, str):
return self.string == other
else:
raise TypeError(
"A token can only be compared to another token or to a string."
)
[docs]
def __repr__(self):
"""Nicely formatted token to help with debugging session.
Note that it does **not** print a string representation that could be
used to create a new ``Token`` instance, which is something you should
never need to do other than indirectly by using the functions
provided in this module.
"""
return _token_format.format(
type="%s (%s)" % (self.type, py_tokenize.tok_name[self.type]),
string=repr(self.string),
start=str(self.start),
end=str(self.end),
line=repr(self.line),
)
[docs]
def __str__(self):
"""Returns the string attribute."""
return self.string
[docs]
def __contains__(self, str_arg):
"""Returns True if the string argument is a substring of the token string attribute"""
if not isinstance(str_arg, str):
return False
return str_arg in self.string
[docs]
def __len__(self):
"""Returns the length of the string attribute"""
return len(self.string)
[docs]
def is_complex(self):
"""Returns True if the token represents a complex number.cavie"""
return self.is_number() and isinstance(ast.literal_eval(self.string), complex)
[docs]
def is_f_string(self):
"""Return True if the token is an f-string"""
return self.type == py_tokenize.STRING and (
self.string.startswith("f") or self.string.startswith("F")
)
[docs]
def is_float(self):
"""Returns True if the token represents a float."""
return self.is_number() and isinstance(ast.literal_eval(self.string), float)
[docs]
def is_identifier(self):
"""Returns ``True`` if the token represents a valid Python identifier
excluding Python keywords.
Note: this is different from Python's string method ``isidentifier``
which also returns ``True`` if the string is a keyword.
"""
return self.string.isidentifier() and not self.is_keyword()
[docs]
def is_in(self, sequence_of_strings):
"""Returns True if the token string is found in the sequence
of strings."""
return self.string in sequence_of_strings
[docs]
def is_indentation(self):
"""Returns True if the token indicates a change in indentation,
(``INDENT``, ``DEDENT``, ``BAD_DEDENT``).
"""
return self.type in (
py_tokenize.INDENT,
py_tokenize.DEDENT,
py_tokenize.BAD_DEDENT,
)
[docs]
def is_integer(self):
"""Returns True if the token represents an integer"""
return self.is_number() and isinstance(ast.literal_eval(self.string), int)
[docs]
def is_keyword(self):
"""Returns True if the token represents a Python keyword."""
return keyword.iskeyword(self.string)
[docs]
def is_name(self):
"""Returns ``True`` if the token is a type NAME"""
return self.type == py_tokenize.NAME
[docs]
def is_newline(self):
"""Returns True if the token type is either ``NEWLINE`` or ``NL``."""
return self.type in (py_tokenize.NEWLINE, py_tokenize.NL)
[docs]
def is_number(self):
"""Returns True if the token represents a number."""
return self.type == py_tokenize.NUMBER
[docs]
def is_operator(self) -> bool:
"""Returns true if the token is of type OP"""
return self.type == py_tokenize.OP
[docs]
def is_space(self):
"""Returns True if the token indicates a change in indentation,
the end of a line, or the end of the source
(``INDENT``, ``DEDENT``, ``BAD_DEDENT``, ``NEWLINE``,
``NL``, and ``ENDMARKER``).
Note that spaces, including tab characters ``\\t``, between tokens
on a given line are not considered to be tokens themselves.
"""
return self.type in (
py_tokenize.INDENT,
py_tokenize.DEDENT,
py_tokenize.BAD_DEDENT,
py_tokenize.NEWLINE,
py_tokenize.NL,
py_tokenize.ENDMARKER,
)
[docs]
def is_string(self):
"""Returns True if the token represents a string"""
return self.type == py_tokenize.STRING
[docs]
def is_unclosed_string(self):
"""Returns True if the token is an unclosed string"""
return self.type in (
py_tokenize.UNCL_SINGLE,
py_tokenize.UNCL_TRIPLE,
)
[docs]
def make_fake_token(
type=py_tokenize.FAKE_TOKEN, string="$", start=(0, 0), end=(0, 0), line=""
):
"""Useful when we need to process a list of tokens with
multiple consecutive at a time, and we need to lengthen
the list for doing so.
Do not use as token to be inserted in a list of tokens
to be untokenize as it will almost certainly not lead to
the desired result. If needed for modifying a list of token
prior to untokenizing, simply insert regular strings instead
of fake tokens.
"""
fake = (type, string, start, end, line)
return Token(fake)