"""
tokenizing.py
==============
All the functions dealing with tokenizing/untokenizing.
"""
from collections import deque
from itertools import chain, islice
from token_utils import _py_tokenize
from io import StringIO
from token_utils.token_class import Token, make_fake_token
def _fix_empty_line(source, prev_token, last_token):
"""Our tokenizer is based on Python's 3.11 tokenizer
which drops entirely a last line if it consists only of
space characters and/or tab characters.
To ensure that we can always have::
untokenize(tokenize(source)) == source
we correct the last token content if needed.
"""
if prev_token is None:
print("WARNING: _fix_empty_line was called with prev_token==None.")
print("This should never happen. Please file an issue and include")
print("the source that produced this result.")
return last_token
if prev_token.line.endswith((" ", "\t")): # fix not needed
return last_token
new_chars = []
for char in reversed(source):
if char in (" ", "\t"):
new_chars.insert(0, char)
else:
break
last_token.string = "".join(new_chars)
return last_token
[docs]
def generate_tokens(source):
"""Tokenize a source (string) yielding tokens one at a time."""
# Unfortunately, there is still a bug in py_tokenize._tokenize
# that drops entirely a last line
# if it consists only of space characters and/or tab characters.
# So, we keep watch for the last token (ENDMARKER) and
# apply a fix if needed.
prev_token = None
perhaps_fix_needed = source.endswith((" ", "\t"))
try:
for tok in _py_tokenize.generate_tokens(StringIO(source).readline):
token = Token(tok)
if token.type != _py_tokenize.ENDMARKER or not perhaps_fix_needed:
yield token
else:
if not source.strip(): # We were passed a useless string!
token.string = source
yield token
else:
yield _fix_empty_line(source, prev_token, token)
prev_token = token
except Exception as exc:
print(
"WARNING: the following unexpected error was raised in ",
"token_utils' generate_tokens().\n",
"Please report this as an issue, including the source that produced this.",
)
print(exc, repr(exc))
[docs]
def tokenize(source, warning=True):
"""Transforms a source (string) into a list of Tokens."""
return list(generate_tokens(source))
[docs]
def get_significant_tokens(source):
"""Gets a list of tokens from a source (str), removing
any token that signal a change in indentation.
"""
# This is useful when we want to analyze the content of a
# line and want to identify a first 'significant' token
# as it will be the first of the line!
tokens = []
for token in generate_tokens(source):
if token.is_indentation():
continue
tokens.append(token)
return tokens
[docs]
def get_physical_lines(source):
"""Transforms a source (string) into a list of of list of Tokens,
with each (inner) list containing all the tokens found on a given
physical line of code.
"""
lines = []
current_row = -1
new_line = []
for token in generate_tokens(source):
if token.start_row != current_row:
current_row = token.start_row
if new_line:
lines.append(new_line)
new_line = []
new_line.append(token)
if new_line:
lines.append(new_line)
return lines
[docs]
def get_stripped_lines(source):
"""Transforms a source (string) into a list of of list of Tokens,
with each (inner) list containing all the tokens found on a given
line of code except that any token related to change in
indentation and comments will have been removed. Thus, for a given
(inner) list of tokens, list[0] will be a non-space token.
"""
lines = []
current_row = -1
new_line = []
first_token = ""
for token in generate_tokens(source):
if token.start_row != current_row:
current_row = token.start_row
if new_line:
lines.append(new_line)
elif first_token:
lines.append([first_token])
new_line = []
first_token = token
if not (token.is_indentation() or token.is_comment()):
new_line.append(token)
if new_line:
lines.append(new_line)
return lines
[docs]
def untokenize_lines_of_tokens(lines):
"""Given a line of lines of tokens, such as that
obtained by ``get_physical_lines()`` or ``get_stripped_lines``,
returns a string containing the source.
The following should be true::
untokenize_lines_of_tokens(get_physical_lines(source)) == source
"""
tokens = [token for line in lines for token in line]
return untokenize(tokens)
[docs]
def untokenize(tokens):
"""Return source code based on tokens.
Adapted from https://github.com/myint/untokenize,
Copyright (C) 2013-2018 Steven Myint, MIT License (same as this project).
This is similar to Python's own tokenize.untokenize(), except that it
preserves spacing between tokens, by using the line
information recorded by Python's tokenize.generate_tokens.
As a result, if the original soure code had multiple spaces between
some tokens or if escaped newlines were used or if tab characters
were present in the original source, those will also be present
in the source code produced by untokenize.
Thus ``source == untokenize(tokenize(source))``.
Note: if you you modifying tokens from an original source:
Instead of full token object, ``untokenize`` will accept simple
strings; however, it will only insert them *as is* without taking them
into account when it comes with figuring out spacing between tokens.
It is often more effective to "mutate" a token string to insert new
content.
Finally, while token_utils only deals with sources as string,
and doesn't do encoding, this function will drop tokens
identified as being of type ``ENCODING``, which would mean that they
came from another source.
"""
# Main changes from original:
# 1. We accumulate substrings in a list, rather than concatenating them
# as we go along
# 2. We allow the inclusion of pure strings as token, but without
# taking their length into consideration.
# Begin code (for extraction by Sphinx)
words = []
previous_line = ""
last_row = 0
last_column = -1
last_non_whitespace_token_type = None
for token in tokens:
if isinstance(token, str):
words.append(token)
continue
if token.type == _py_tokenize.ENCODING:
continue
# Preserve escaped newlines.
if (
last_non_whitespace_token_type != _py_tokenize.COMMENT
and token.start_row > last_row
and previous_line.endswith(("\\\n", "\\\r\n", "\\\r"))
):
words.append(previous_line[len(previous_line.rstrip(" \t\n\r\\")) :])
# Preserve spacing.
if token.start_row > last_row:
last_column = 0
if token.start_col > last_column:
# Insert the content that was skipped between tokens
words.append(token.line[last_column : token.start_col])
words.append(token.string)
previous_line = token.line
last_row = token.end_row
last_column = token.end_col
if not token.is_space():
last_non_whitespace_token_type = token.type
return "".join(words)
# End code (for extraction by Sphinx)
[docs]
def stringify(tokens, remove_comments=False):
"""Returns a string built from tokens.
It is somewhat similar to untokenize except that it doesn't add any
missing information from tokens that might have been removed,
inserting spaces instead. For example, removing "two"::
|one two three| -->
|one three|
If no token has been removed from a tokenized list, and no
tab characters are used for indentation or spacing between tokens,
it should return the same content as untokenize.
It allows for easy removal of comments with ``remove_comments=True``.
As long as one does not care about tab characters, it is slightly more
efficient to use::
stringify(tokenize(source), remove_comments=True)
than::
ideas.utils.remove_comments(source)
"""
words = []
previous_line = ""
last_row = 0
last_column = -1
last_non_whitespace_token_type = None
for token in tokens:
if isinstance(token, str):
words.append(token)
continue
if token.type == _py_tokenize.ENCODING:
continue
if remove_comments and token.is_comment():
continue
# Preserve escaped newlines.
if (
not remove_comments
and last_non_whitespace_token_type != _py_tokenize.COMMENT
and token.start_row > last_row
and previous_line.endswith(("\\\n", "\\\r\n", "\\\r"))
):
words.append(previous_line[len(previous_line.rstrip(" \t\n\r\\")) :])
# Preserve spacing.
if token.start_row > last_row:
last_column = 0
if token.start_col > last_column and not token.is_space():
# Insert spaces instead of the content that was skipped between tokens
# except at the end of line before a NL or NEWLINE token
words.append(" " * (token.start_col - last_column))
words.append(token.string)
previous_line = token.line
last_row = token.end_row
last_column = token.end_col
if not token.is_space():
last_non_whitespace_token_type = token.type
return "".join(words)
[docs]
def print_tokens(source):
"""Prints tokens found in source, excluding spaces and comments.
``source`` is either a string to be tokenized, or a list of Token objects.
This is occasionally useful as a debugging tool.
"""
if isinstance(source[0], Token):
source = untokenize(source)
for lines in get_physical_lines(source):
for token in lines:
print(repr(token))
print()
[docs]
def pairwise(iterable, prev=0):
"""Similar to itertools.pairwise (Python 3.10+). However, it adds a fake
token at the end if ``prev==0`` (the default) or at the beginning
if ``prev==1``. Any other value will result in a ``ValueError``
Given a list of tokens represented by lower case letters, and a fake token by F,
the default corresponds to something like::
pairwise('abcde') → Fa ab bc cd de
"""
if prev not in [0, 1]:
raise ValueError("'prev' must be either 0 (the default) or 1.")
iterator = iter(iterable)
if prev:
updated_iterator = chain([make_fake_token()], iterator)
else:
updated_iterator = chain(iterator, [make_fake_token()])
a = next(updated_iterator, None)
for b in updated_iterator:
yield a, b
a = b
[docs]
def sliding_window(iterable, n, prev=0):
"""Collect data into overlapping fixed-length chunks or blocks.
This is inspired by an itertools recipe. Given an iterator,
and a requested 'window' of size 'n', by default (prev=0),
it adds 'n-1' fake token at the end and iterates, emitting 'n'
items at a time until all the items have been served.
Given a list of tokens represented by lower case letters,
and F representing a fake token, we would have something like:
sliding_window('abcde', 3) → abc bcd cde deF eFF
With ``prev==1``, we would have 1 fake token prepended and
one appended; thus
sliding_window('abcde', 3, prev=1) → Fab abc bcd cde deF
We must have ``0 <= prev < n``, otherwise a ValueError is raised
We essentially have::
sliding_window(iterable, 2) == pairwise(iterator)
"""
if not 0 <= prev < n:
raise ValueError(f"'prev' must be in the interval [0, {n}]")
iterator = iter(iterable)
fake = [make_fake_token(string=f"FAKE_{i}") for i in range(n - 1)]
iterator = iter(iterable)
updated_iterator = chain(fake[:prev], iterator, fake[prev : n - 1])
window = deque(islice(updated_iterator, n - 1), maxlen=n)
for x in updated_iterator:
window.append(x)
yield tuple(window)
[docs]
def get_number_significant_tokens(tokens):
"""Given a list of tokens representing a single line,
gives a count of the number of tokens which are not space tokens
(such as ``NEWLINE``, ``INDENT``, ``DEDENT``, etc.) nor
comments.
If the list of tokens includes tokens from more than a single
line, an exception is raised.
"""
if len(tokens) == 0:
raise ValueError("At least one token must be included in the list.")
row = tokens[0].start_row
nb = 0
for token in tokens:
if token.start_row != row:
raise ValueError(
"The list of tokens includes tokens from more "
+ "than one line of code."
)
if token.is_space() or token.is_comment():
continue
nb += 1
return nb
[docs]
def dedent(tokens, nb):
"""Given a list of tokens representing a line,
produces an equivalent list corresponding
to a line of code with the first nb characters removed.
If the list includes tokens from more than one line,
or no token at all, a ``ValueError`` is raised.
If an attempt to remove non-space characters is made,
a ``TypeError`` is raised.
If a negative value for nb is used, the line is indented
by spaces or tab characters instead, with the
first character determining if spaces or tab characters
must be used.
"""
# The "indent" part is probably not needed...
if len(tokens) == 0:
raise ValueError("Empty list of tokens was passed to dedent()/indent().")
row = tokens[0].start_row
for token in tokens:
if token.start_row != row:
raise ValueError(
"Tokens in dedent()/indent() do not come from a single line of code."
)
line = untokenize(tokens)
if nb >= 0:
begin = line[:nb]
end = line[nb:]
if begin.strip():
raise TypeError("Attempting to remove non-space character in dedent().")
return tokenize(end)
nb = -nb
if len(line) == 0:
return tokenize(" " * nb)
first_char = line[0]
if first_char == "\t":
line = "\t" * nb + line
else:
line = " " * nb + line
return tokenize(line)
[docs]
def indent(tokens, n):
"""Calls dedent(tokens, -n) and adds the required number of spaces
or tab characters as needed"""
return dedent(tokens, -n)
[docs]
class BracketStack:
"""Helpful in keeping track of open and close brackets in a token sequence.
It is intended to be strict, and only accept Token instances, with open brackets
added before any close one can be added.
"""
def __init__(self):
self.stack = []
[docs]
def add(self, bracket):
"""Adds a bracket to a stack.
If it is a closing bracket that matches the last added one,
that last opening one is returned.
If it is a close bracket not matching the last added one, or the
first one added, a TypeError is raised.
If an open bracket is added, False is returned.
"""
assert isinstance(bracket, Token)
if self.stack:
if bracket.is_matching_bracket(self.stack[-1]):
return self.stack.pop()
elif not bracket.is_open_bracket():
raise TypeError(
f"Close bracket does not match anything: {repr(bracket)}"
)
else:
self.stack.append(bracket)
return False
elif not bracket.is_open_bracket():
raise TypeError(
f"First bracket added cannot be a closing one: {repr(bracket)}"
)
else:
self.stack.append(bracket)
return False
[docs]
def is_empty(self):
"""Return True if the stack is empty, False if it contains brackets."""
return not bool(self.stack)
__all__ = ["__all__"]
_names = dir()
def _make_all():
for name in _names:
if not name.startswith("_"):
__all__.append(name)
_make_all()
__all__.remove("__all__")
__all__.remove("Token")
__all__.remove("StringIO")
__all__.remove("chain")
__all__.remove("islice")
__all__.remove("deque")