Source code for token_utils.tokenizing

"""
tokenizing.py
==============

All the functions dealing with tokenizing/untokenizing.
"""

from collections import deque
from itertools import chain, islice

from token_utils import _py_tokenize

from io import StringIO

from token_utils.token_class import Token, make_fake_token


def _fix_empty_line(source, prev_token, last_token):
    """Our tokenizer is based on Python's 3.11 tokenizer
    which drops entirely a last line if it consists only of
    space characters and/or tab characters.

    To ensure that we can always have::

        untokenize(tokenize(source)) == source

    we correct the last token content if needed.
    """
    if prev_token is None:
        print("WARNING: _fix_empty_line was called with prev_token==None.")
        print("This should never happen. Please file an issue and include")
        print("the source that produced this result.")
        return last_token

    if prev_token.line.endswith((" ", "\t")):  # fix not needed
        return last_token

    new_chars = []
    for char in reversed(source):
        if char in (" ", "\t"):
            new_chars.insert(0, char)
        else:
            break
    last_token.string = "".join(new_chars)
    return last_token


[docs] def generate_tokens(source): """Tokenize a source (string) yielding tokens one at a time.""" # Unfortunately, there is still a bug in py_tokenize._tokenize # that drops entirely a last line # if it consists only of space characters and/or tab characters. # So, we keep watch for the last token (ENDMARKER) and # apply a fix if needed. prev_token = None perhaps_fix_needed = source.endswith((" ", "\t")) try: for tok in _py_tokenize.generate_tokens(StringIO(source).readline): token = Token(tok) if token.type != _py_tokenize.ENDMARKER or not perhaps_fix_needed: yield token else: if not source.strip(): # We were passed a useless string! token.string = source yield token else: yield _fix_empty_line(source, prev_token, token) prev_token = token except Exception as exc: print( "WARNING: the following unexpected error was raised in ", "token_utils' generate_tokens().\n", "Please report this as an issue, including the source that produced this.", ) print(exc, repr(exc))
[docs] def tokenize(source, warning=True): """Transforms a source (string) into a list of Tokens.""" return list(generate_tokens(source))
[docs] def get_significant_tokens(source): """Gets a list of tokens from a source (str), removing any token that signal a change in indentation. """ # This is useful when we want to analyze the content of a # line and want to identify a first 'significant' token # as it will be the first of the line! tokens = [] for token in generate_tokens(source): if token.is_indentation(): continue tokens.append(token) return tokens
[docs] def get_physical_lines(source): """Transforms a source (string) into a list of of list of Tokens, with each (inner) list containing all the tokens found on a given physical line of code. """ lines = [] current_row = -1 new_line = [] for token in generate_tokens(source): if token.start_row != current_row: current_row = token.start_row if new_line: lines.append(new_line) new_line = [] new_line.append(token) if new_line: lines.append(new_line) return lines
[docs] def get_stripped_lines(source): """Transforms a source (string) into a list of of list of Tokens, with each (inner) list containing all the tokens found on a given line of code except that any token related to change in indentation and comments will have been removed. Thus, for a given (inner) list of tokens, list[0] will be a non-space token. """ lines = [] current_row = -1 new_line = [] first_token = "" for token in generate_tokens(source): if token.start_row != current_row: current_row = token.start_row if new_line: lines.append(new_line) elif first_token: lines.append([first_token]) new_line = [] first_token = token if not (token.is_indentation() or token.is_comment()): new_line.append(token) if new_line: lines.append(new_line) return lines
[docs] def untokenize_lines_of_tokens(lines): """Given a line of lines of tokens, such as that obtained by ``get_physical_lines()`` or ``get_stripped_lines``, returns a string containing the source. The following should be true:: untokenize_lines_of_tokens(get_physical_lines(source)) == source """ tokens = [token for line in lines for token in line] return untokenize(tokens)
[docs] def untokenize(tokens): """Return source code based on tokens. Adapted from https://github.com/myint/untokenize, Copyright (C) 2013-2018 Steven Myint, MIT License (same as this project). This is similar to Python's own tokenize.untokenize(), except that it preserves spacing between tokens, by using the line information recorded by Python's tokenize.generate_tokens. As a result, if the original soure code had multiple spaces between some tokens or if escaped newlines were used or if tab characters were present in the original source, those will also be present in the source code produced by untokenize. Thus ``source == untokenize(tokenize(source))``. Note: if you you modifying tokens from an original source: Instead of full token object, ``untokenize`` will accept simple strings; however, it will only insert them *as is* without taking them into account when it comes with figuring out spacing between tokens. It is often more effective to "mutate" a token string to insert new content. Finally, while token_utils only deals with sources as string, and doesn't do encoding, this function will drop tokens identified as being of type ``ENCODING``, which would mean that they came from another source. """ # Main changes from original: # 1. We accumulate substrings in a list, rather than concatenating them # as we go along # 2. We allow the inclusion of pure strings as token, but without # taking their length into consideration. # Begin code (for extraction by Sphinx) words = [] previous_line = "" last_row = 0 last_column = -1 last_non_whitespace_token_type = None for token in tokens: if isinstance(token, str): words.append(token) continue if token.type == _py_tokenize.ENCODING: continue # Preserve escaped newlines. if ( last_non_whitespace_token_type != _py_tokenize.COMMENT and token.start_row > last_row and previous_line.endswith(("\\\n", "\\\r\n", "\\\r")) ): words.append(previous_line[len(previous_line.rstrip(" \t\n\r\\")) :]) # Preserve spacing. if token.start_row > last_row: last_column = 0 if token.start_col > last_column: # Insert the content that was skipped between tokens words.append(token.line[last_column : token.start_col]) words.append(token.string) previous_line = token.line last_row = token.end_row last_column = token.end_col if not token.is_space(): last_non_whitespace_token_type = token.type return "".join(words)
# End code (for extraction by Sphinx)
[docs] def stringify(tokens, remove_comments=False): """Returns a string built from tokens. It is somewhat similar to untokenize except that it doesn't add any missing information from tokens that might have been removed, inserting spaces instead. For example, removing "two":: |one two three| --> |one three| If no token has been removed from a tokenized list, and no tab characters are used for indentation or spacing between tokens, it should return the same content as untokenize. It allows for easy removal of comments with ``remove_comments=True``. As long as one does not care about tab characters, it is slightly more efficient to use:: stringify(tokenize(source), remove_comments=True) than:: ideas.utils.remove_comments(source) """ words = [] previous_line = "" last_row = 0 last_column = -1 last_non_whitespace_token_type = None for token in tokens: if isinstance(token, str): words.append(token) continue if token.type == _py_tokenize.ENCODING: continue if remove_comments and token.is_comment(): continue # Preserve escaped newlines. if ( not remove_comments and last_non_whitespace_token_type != _py_tokenize.COMMENT and token.start_row > last_row and previous_line.endswith(("\\\n", "\\\r\n", "\\\r")) ): words.append(previous_line[len(previous_line.rstrip(" \t\n\r\\")) :]) # Preserve spacing. if token.start_row > last_row: last_column = 0 if token.start_col > last_column and not token.is_space(): # Insert spaces instead of the content that was skipped between tokens # except at the end of line before a NL or NEWLINE token words.append(" " * (token.start_col - last_column)) words.append(token.string) previous_line = token.line last_row = token.end_row last_column = token.end_col if not token.is_space(): last_non_whitespace_token_type = token.type return "".join(words)
[docs] def pairwise(iterable, prev=0): """Similar to itertools.pairwise (Python 3.10+). However, it adds a fake token at the end if ``prev==0`` (the default) or at the beginning if ``prev==1``. Any other value will result in a ``ValueError`` Given a list of tokens represented by lower case letters, and a fake token by F, the default corresponds to something like:: pairwise('abcde') → Fa ab bc cd de """ if prev not in [0, 1]: raise ValueError("'prev' must be either 0 (the default) or 1.") iterator = iter(iterable) if prev: updated_iterator = chain([make_fake_token()], iterator) else: updated_iterator = chain(iterator, [make_fake_token()]) a = next(updated_iterator, None) for b in updated_iterator: yield a, b a = b
[docs] def sliding_window(iterable, n, prev=0): """Collect data into overlapping fixed-length chunks or blocks. This is inspired by an itertools recipe. Given an iterator, and a requested 'window' of size 'n', by default (prev=0), it adds 'n-1' fake token at the end and iterates, emitting 'n' items at a time until all the items have been served. Given a list of tokens represented by lower case letters, and F representing a fake token, we would have something like: sliding_window('abcde', 3) → abc bcd cde deF eFF With ``prev==1``, we would have 1 fake token prepended and one appended; thus sliding_window('abcde', 3, prev=1) → Fab abc bcd cde deF We must have ``0 <= prev < n``, otherwise a ValueError is raised We essentially have:: sliding_window(iterable, 2) == pairwise(iterator) """ if not 0 <= prev < n: raise ValueError(f"'prev' must be in the interval [0, {n}]") iterator = iter(iterable) fake = [make_fake_token(string=f"FAKE_{i}") for i in range(n - 1)] iterator = iter(iterable) updated_iterator = chain(fake[:prev], iterator, fake[prev : n - 1]) window = deque(islice(updated_iterator, n - 1), maxlen=n) for x in updated_iterator: window.append(x) yield tuple(window)
[docs] def get_number_significant_tokens(tokens): """Given a list of tokens representing a single line, gives a count of the number of tokens which are not space tokens (such as ``NEWLINE``, ``INDENT``, ``DEDENT``, etc.) nor comments. If the list of tokens includes tokens from more than a single line, an exception is raised. """ if len(tokens) == 0: raise ValueError("At least one token must be included in the list.") row = tokens[0].start_row nb = 0 for token in tokens: if token.start_row != row: raise ValueError( "The list of tokens includes tokens from more " + "than one line of code." ) if token.is_space() or token.is_comment(): continue nb += 1 return nb
[docs] def strip_comments(source): """Removes the comments in a source. It also removes any space at the end of each line (before the ``\\n`` if present). """ # The untokenizing function uses not only the string attribute # but also the start_col, end_col, and line attributes # to see if any character included in the line attribute # between the end_col of a token preceeding the start_col # of another must be included. Thus, we must not simply remove # tokens from a stream unless they contain only spaces, # otherwise we might not get the desired result. tokens = [] for token in generate_tokens(source): if token.is_comment(): token.string = "" # does not remove any space preceeding it. tokens.append(token) mid_removal = untokenize(tokens) # now we remove the extra spaces before the commment lines = mid_removal.split("\n") new_lines = [line.rstrip() for line in lines] return "\n".join(new_lines)
[docs] def dedent(tokens, nb): """Given a list of tokens representing a line, produces an equivalent list corresponding to a line of code with the first nb characters removed. If the list includes tokens from more than one line, or no token at all, a ``ValueError`` is raised. If an attempt to remove non-space characters is made, a ``TypeError`` is raised. If a negative value for nb is used, the line is indented by spaces or tab characters instead, with the first character determining if spaces or tab characters must be used. """ # The "indent" part is probably not needed... if len(tokens) == 0: raise ValueError("Empty list of tokens was passed to dedent()/indent().") row = tokens[0].start_row for token in tokens: if token.start_row != row: raise ValueError( "Tokens in dedent()/indent() do not come from a single line of code." ) line = untokenize(tokens) if nb >= 0: begin = line[:nb] end = line[nb:] if begin.strip(): raise TypeError("Attempting to remove non-space character in dedent().") return tokenize(end) nb = -nb if len(line) == 0: return tokenize(" " * nb) first_char = line[0] if first_char == "\t": line = "\t" * nb + line else: line = " " * nb + line return tokenize(line)
[docs] def indent(tokens, n): """Calls dedent(tokens, -n) and adds the required number of spaces or tab characters as needed""" return dedent(tokens, -n)
[docs] class BracketStack: """Helpful in keeping track of open and close brackets in a token sequence. It is intended to be strict, and only accept Token instances, with open brackets added before any close one can be added. """ def __init__(self): self.stack = []
[docs] def add(self, bracket): """Adds a bracket to a stack. If it is a closing bracket that matches the last added one, that last opening one is returned. If it is a close bracket not matching the last added one, or the first one added, a TypeError is raised. If an open bracket is added, False is returned. """ assert isinstance(bracket, Token) if self.stack: if bracket.is_matching_bracket(self.stack[-1]): return self.stack.pop() elif not bracket.is_open_bracket(): raise TypeError( f"Close bracket does not match anything: {repr(bracket)}" ) else: self.stack.append(bracket) return False elif not bracket.is_open_bracket(): raise TypeError( f"First bracket added cannot be a closing one: {repr(bracket)}" ) else: self.stack.append(bracket) return False
[docs] def is_empty(self): """Return True if the stack is empty, False if it contains brackets.""" return not bool(self.stack)
__all__ = ["__all__"] _names = dir() def _make_all(): for name in _names: if not name.startswith("_"): __all__.append(name) _make_all() __all__.remove("__all__") __all__.remove("Token") __all__.remove("StringIO") __all__.remove("chain") __all__.remove("islice") __all__.remove("deque")