Source code for token_utils.token_class

import ast
import keyword

from token_utils import py_tokenize

_token_format = "type={type}  string={string}  start={start}  end={end}  line={line}"


[docs] class Token: """Token as generated from Python's tokenize.generate_tokens written here in a more convenient form, and with some custom methods. The various parameters are:: type: token type string: the token written as a string start = (start_row, start_col) end = (end_row, end_col) line: entire line of code where the token is found. Token instances are mutable objects. Therefore, given a list of tokens, we can change the value of any token's attribute, untokenize the list and automatically obtain a transformed source. """
[docs] def __init__(self, token): """Initializes using a token produced by Python's tokenize function as input.""" self.type = token[0] self.string = token[1] self.start = self.start_row, self.start_col = token[2] self.end = self.end_row, self.end_col = token[3] self.line = token[4]
[docs] def __eq__(self, other): """Compares a Token with another object; returns true if self.string == other.string or if self.string == other. """ if hasattr(other, "string"): return self.string == other.string elif isinstance(other, str): return self.string == other else: raise TypeError( "A token can only be compared to another token or to a string." )
[docs] def __repr__(self): """Nicely formatted token to help with debugging session. Note that it does **not** print a string representation that could be used to create a new ``Token`` instance, which is something you should never need to do other than indirectly by using the functions provided in this module. """ return _token_format.format( type="%s (%s)" % (self.type, py_tokenize.tok_name[self.type]), string=repr(self.string), start=str(self.start), end=str(self.end), line=repr(self.line), )
[docs] def __str__(self): """Returns the string attribute.""" return self.string
[docs] def __contains__(self, str_arg): """Returns True if the string argument is a substring of the token string attribute""" if not isinstance(str_arg, str): return False return str_arg in self.string
[docs] def __len__(self): """Returns the length of the string attribute""" return len(self.string)
[docs] def is_comment(self): """Returns True if the token is a comment.""" return self.type == py_tokenize.COMMENT
[docs] def is_complex(self): """Returns True if the token represents a complex number.cavie""" return self.is_number() and isinstance(ast.literal_eval(self.string), complex)
[docs] def is_f_string(self): """Return True if the token is an f-string""" return self.type == py_tokenize.STRING and ( self.string.startswith("f") or self.string.startswith("F") )
[docs] def is_float(self): """Returns True if the token represents a float.""" return self.is_number() and isinstance(ast.literal_eval(self.string), float)
[docs] def is_identifier(self): """Returns ``True`` if the token represents a valid Python identifier excluding Python keywords. Note: this is different from Python's string method ``isidentifier`` which also returns ``True`` if the string is a keyword. """ return self.string.isidentifier() and not self.is_keyword()
[docs] def is_immediately_before(self, other): """Returns True if the current token is immediately before other, without any intervening space in between the two tokens. """ if not isinstance(other, Token): # pragma: no cover return False return self.end_row == other.start_row and self.end_col == other.start_col
[docs] def is_immediately_after(self, other): """Returns True if the current token is immediately after other, without any intervening space in between the two tokens. """ if not isinstance(other, Token): # pragma: no cover return False return other.is_immediately_before(self)
[docs] def is_in(self, sequence_of_strings): """Returns True if the token string is found in the sequence of strings.""" return self.string in sequence_of_strings
[docs] def is_indentation(self): """Returns True if the token indicates a change in indentation, (``INDENT``, ``DEDENT``, ``BAD_DEDENT``). """ return self.type in ( py_tokenize.INDENT, py_tokenize.DEDENT, py_tokenize.BAD_DEDENT, )
[docs] def is_integer(self): """Returns True if the token represents an integer""" return self.is_number() and isinstance(ast.literal_eval(self.string), int)
[docs] def is_keyword(self): """Returns True if the token represents a Python keyword.""" return keyword.iskeyword(self.string)
[docs] def is_name(self): """Returns ``True`` if the token is a type NAME""" return self.type == py_tokenize.NAME
[docs] def is_newline(self): """Returns True if the token type is either ``NEWLINE`` or ``NL``.""" return self.type in (py_tokenize.NEWLINE, py_tokenize.NL)
[docs] def is_number(self): """Returns True if the token represents a number.""" return self.type == py_tokenize.NUMBER
[docs] def is_operator(self) -> bool: """Returns true if the token is of type OP""" return self.type == py_tokenize.OP
[docs] def is_space(self): """Returns True if the token indicates a change in indentation, the end of a line, or the end of the source (``INDENT``, ``DEDENT``, ``BAD_DEDENT``, ``NEWLINE``, ``NL``, and ``ENDMARKER``). Note that spaces, including tab characters ``\\t``, between tokens on a given line are not considered to be tokens themselves. """ return self.type in ( py_tokenize.INDENT, py_tokenize.DEDENT, py_tokenize.BAD_DEDENT, py_tokenize.NEWLINE, py_tokenize.NL, py_tokenize.ENDMARKER, )
[docs] def is_string(self): """Returns True if the token represents a string""" return self.type == py_tokenize.STRING
[docs] def is_unclosed_string(self): """Returns True if the token is an unclosed string""" return self.type in ( py_tokenize.UNCL_SINGLE, py_tokenize.UNCL_TRIPLE, )
[docs] def make_fake_token( type=py_tokenize.FAKE_TOKEN, string="$", start=(0, 0), end=(0, 0), line="" ): """Useful when we need to process a list of tokens with multiple consecutive at a time, and we need to lengthen the list for doing so. Do not use as token to be inserted in a list of tokens to be untokenize as it will almost certainly not lead to the desired result. If needed for modifying a list of token prior to untokenizing, simply insert regular strings instead of fake tokens. """ fake = (type, string, start, end, line) return Token(fake)