add changes from tclint
This commit is contained in:
+45
-319
@@ -1,3 +1,5 @@
|
|||||||
|
import io
|
||||||
|
from typing import Optional, Tuple
|
||||||
from tclint.parser import Parser
|
from tclint.parser import Parser
|
||||||
from tclint.commands import CommandArgError
|
from tclint.commands import CommandArgError
|
||||||
from tclint.syntax_tree import (
|
from tclint.syntax_tree import (
|
||||||
@@ -8,254 +10,17 @@ from tclint.syntax_tree import (
|
|||||||
QuotedWord,
|
QuotedWord,
|
||||||
Expression,
|
Expression,
|
||||||
)
|
)
|
||||||
|
from tclint.lexer import TclSyntaxError, Lexer, TOK_EOF
|
||||||
import ply.lex as lex
|
|
||||||
from typing import Tuple
|
|
||||||
|
|
||||||
TOK_BACKSLASH_NEWLINE = "BACKSLASH_NEWLINE"
|
|
||||||
TOK_BACKSLASH_SUB = "BACKSLASH_SUB"
|
|
||||||
TOK_NEWLINE = "NEWLINE"
|
|
||||||
TOK_SEMI = "SEMI"
|
|
||||||
TOK_WS = "WS"
|
|
||||||
TOK_QUOTE = "QUOTE"
|
|
||||||
TOK_ARG_EXPANSION = "ARG_EXPANSION"
|
|
||||||
TOK_LBRACE = "LBRACE"
|
|
||||||
TOK_RBRACE = "RBRACE"
|
|
||||||
TOK_STAR = "STAR"
|
|
||||||
TOK_LBRACKET = "LBRACKET"
|
|
||||||
TOK_RBRACKET = "RBRACKET"
|
|
||||||
TOK_DOLLAR = "DOLLAR"
|
|
||||||
TOK_LPAREN = "LPAREN"
|
|
||||||
TOK_RPAREN = "RPAREN"
|
|
||||||
TOK_HASH = "HASH"
|
|
||||||
TOK_ALPHA_CHARS = "ALPHA_CHARS"
|
|
||||||
TOK_NUM_CHARS = "NUM_CHARS"
|
|
||||||
TOK_NAMESPACE_SEP = "NAMESPACE_SEP"
|
|
||||||
TOK_CHAR = "CHAR"
|
|
||||||
TOK_CONTENTS = "CONTENTS"
|
|
||||||
TOK_EOF = None
|
|
||||||
|
|
||||||
STATE_BRACEDWORD = "bracedword"
|
|
||||||
|
|
||||||
|
|
||||||
class TclSyntaxError(Exception):
|
|
||||||
def __init__(self, message, start: Tuple[int, int], end: Tuple[int, int]):
|
|
||||||
super().__init__(message)
|
|
||||||
self.start = start
|
|
||||||
self.end = end
|
|
||||||
|
|
||||||
|
|
||||||
class _LexTable:
|
|
||||||
tokens = (
|
|
||||||
TOK_BACKSLASH_NEWLINE,
|
|
||||||
TOK_BACKSLASH_SUB,
|
|
||||||
TOK_NEWLINE,
|
|
||||||
TOK_SEMI,
|
|
||||||
TOK_WS,
|
|
||||||
TOK_QUOTE,
|
|
||||||
TOK_ARG_EXPANSION,
|
|
||||||
TOK_LBRACE,
|
|
||||||
TOK_RBRACE,
|
|
||||||
TOK_STAR,
|
|
||||||
TOK_LBRACKET,
|
|
||||||
TOK_RBRACKET,
|
|
||||||
TOK_DOLLAR,
|
|
||||||
TOK_LPAREN,
|
|
||||||
TOK_RPAREN,
|
|
||||||
TOK_HASH,
|
|
||||||
TOK_ALPHA_CHARS,
|
|
||||||
TOK_NUM_CHARS,
|
|
||||||
TOK_NAMESPACE_SEP,
|
|
||||||
TOK_CHAR,
|
|
||||||
TOK_CONTENTS,
|
|
||||||
)
|
|
||||||
|
|
||||||
# This defines a conditional lexing state for parsing braced words. This is a
|
|
||||||
# performance optimization; since there are few special characters in this context,
|
|
||||||
# we can use a smaller set of tokens to parse them faster. This has a large impact
|
|
||||||
# since most Tcl programs have a large number of braced words. Any token with
|
|
||||||
# `bracedword` in its name is included in this state. Tokens that are included in
|
|
||||||
# this state and the default state also include `INITIAL` in their name.
|
|
||||||
states = ((STATE_BRACEDWORD, "exclusive"),)
|
|
||||||
|
|
||||||
def _tok(self, t):
|
|
||||||
pos = (t.lexer.lineno, t.lexer.colno)
|
|
||||||
t.lexer.lineno += t.value.count("\n")
|
|
||||||
index = t.value.rfind("\n")
|
|
||||||
if index == -1:
|
|
||||||
t.lexer.colno += len(t.value)
|
|
||||||
else:
|
|
||||||
remaining = t.value[index + 1 :]
|
|
||||||
t.lexer.colno = len(remaining) + 1
|
|
||||||
|
|
||||||
t.value = (t.value, pos)
|
|
||||||
return t
|
|
||||||
|
|
||||||
# Priority important
|
|
||||||
def t_bracedword_INITIAL_BACKSLASH_NEWLINE(self, t):
|
|
||||||
r"\\\r?\n"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
# Priority important
|
|
||||||
def t_bracedword_INITIAL_BACKSLASH_SUB(self, t):
|
|
||||||
r"\\."
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_NEWLINE(self, t):
|
|
||||||
r"\n"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_SEMI(self, t):
|
|
||||||
r";"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
# TODO: should use \s?
|
|
||||||
def t_WS(self, t):
|
|
||||||
r"[\t\v\f\r ]+"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_QUOTE(self, t):
|
|
||||||
r'"'
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
# Must be higher priority than LBRACE
|
|
||||||
def t_ARG_EXPANSION(self, t):
|
|
||||||
r"\{\*\}"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_bracedword_INITIAL_LBRACE(self, t):
|
|
||||||
r"\{"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_bracedword_INITIAL_RBRACE(self, t):
|
|
||||||
r"\}"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_STAR(self, t):
|
|
||||||
r"\*"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_LBRACKET(self, t):
|
|
||||||
r"\["
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_RBRACKET(self, t):
|
|
||||||
r"\]"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_DOLLAR(self, t):
|
|
||||||
r"\$"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_LPAREN(self, t):
|
|
||||||
r"\("
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_RPAREN(self, t):
|
|
||||||
r"\)"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_HASH(self, t):
|
|
||||||
r"\#"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
# Valid non-numeric chars in variable names
|
|
||||||
def t_ALPHA_CHARS(self, t):
|
|
||||||
r"[A-Za-z_]+"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
# Valid numeric chars in variable names
|
|
||||||
# This is split up from the above to facilitate expression parsing, since
|
|
||||||
# e.g. 1eq1 can't be a single token.
|
|
||||||
def t_NUM_CHARS(self, t):
|
|
||||||
r"[0-9]+"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_NAMESPACE_SEP(self, t):
|
|
||||||
r"::+"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
def t_bracedword_CONTENTS(self, t):
|
|
||||||
r"[^{}\\]+"
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
# Catch-all. TODO: inefficient, should probably munch multiple chars
|
|
||||||
def t_CHAR(self, t):
|
|
||||||
r"."
|
|
||||||
return self._tok(t)
|
|
||||||
|
|
||||||
# Error handling rule
|
|
||||||
# TODO: do we need this? since we have a catch-all...
|
|
||||||
# there is a warning
|
|
||||||
def t_bracedword_INITIAL_error(self, t):
|
|
||||||
print("Illegal character '%s'" % t.value[0])
|
|
||||||
t.lexer.skip(1)
|
|
||||||
|
|
||||||
def __init__(self):
|
|
||||||
self.lexer = lex.lex(object=self)
|
|
||||||
self.lexer.lineno = 1
|
|
||||||
self.lexer.colno = 1
|
|
||||||
|
|
||||||
def new_lexer(self, pos=None):
|
|
||||||
lexer = self.lexer.clone()
|
|
||||||
lexer.lineno = 1
|
|
||||||
lexer.colno = 1
|
|
||||||
|
|
||||||
if pos is not None:
|
|
||||||
line, col = pos
|
|
||||||
lexer.lineno = line
|
|
||||||
lexer.colno = col
|
|
||||||
|
|
||||||
return lexer
|
|
||||||
|
|
||||||
|
|
||||||
# Calling `lex.lex()` performs an expensive reflection process to generate the lexer.
|
|
||||||
# This singleton class holds a preinitialized lexer that can then be cloned to create
|
|
||||||
# individual instances.
|
|
||||||
LexTable = _LexTable()
|
|
||||||
|
|
||||||
|
|
||||||
class Lexer:
|
|
||||||
def __init__(self, pos=None):
|
|
||||||
self.lexer = LexTable.new_lexer(pos)
|
|
||||||
self.current = None
|
|
||||||
|
|
||||||
def input(self, text):
|
|
||||||
self.lexer.input(text)
|
|
||||||
self.current = self.lexer.token()
|
|
||||||
|
|
||||||
def type(self):
|
|
||||||
if self.current is None:
|
|
||||||
return TOK_EOF
|
|
||||||
return self.current.type
|
|
||||||
|
|
||||||
def value(self):
|
|
||||||
if self.current is None:
|
|
||||||
return None
|
|
||||||
return self.current.value[0]
|
|
||||||
|
|
||||||
def pos(self):
|
|
||||||
if self.current is None:
|
|
||||||
return (self.lexer.lineno, self.lexer.colno)
|
|
||||||
return self.current.value[1]
|
|
||||||
|
|
||||||
def next(self):
|
|
||||||
self.current = self.lexer.token()
|
|
||||||
|
|
||||||
def expect(self, *tokens, message, pos):
|
|
||||||
if self.type() not in tokens:
|
|
||||||
self.next() # munch another token to update position
|
|
||||||
raise TclSyntaxError(message, pos, self.pos())
|
|
||||||
|
|
||||||
self.next()
|
|
||||||
|
|
||||||
def assert_(self, *tokens):
|
|
||||||
assert self.current.type in tokens
|
|
||||||
self.next()
|
|
||||||
|
|
||||||
|
|
||||||
class CustomParser(Parser):
|
class CustomParser(Parser):
|
||||||
def parse(self, script, pos=None):
|
def __init__(self, debug=False, command_plugins=None):
|
||||||
|
super().__init__(debug, command_plugins)
|
||||||
|
# Used to normalize newlines consistently with open()'s universal newlines mode.
|
||||||
|
self._decoder = io.IncrementalNewlineDecoder(None, True)
|
||||||
|
|
||||||
|
def parse(self, script: str, pos: Optional[Tuple[int, int]] = None):
|
||||||
|
script = self._decoder.decode(script, True)
|
||||||
lexer = Lexer(pos=pos)
|
lexer = Lexer(pos=pos)
|
||||||
lexer.input(script)
|
lexer.input(script)
|
||||||
tree = self._parse_script(lexer, in_command_sub=False)
|
tree = self._parse_script(lexer, in_command_sub=False)
|
||||||
@@ -265,83 +30,44 @@ class CustomParser(Parser):
|
|||||||
|
|
||||||
return tree
|
return tree
|
||||||
|
|
||||||
def parse_list(self, node):
|
def _parse_operator(self, ts):
|
||||||
"""Parse contents of node as Tcl list. This is a distinct entry point
|
pos = ts.pos()
|
||||||
that doesn't get used when generating the main syntax tree, but is used
|
|
||||||
in command-specific argument parsing.
|
|
||||||
"""
|
|
||||||
if isinstance(node, List):
|
|
||||||
return node
|
|
||||||
|
|
||||||
if node.contents is None:
|
# hacky logic to handle parsing legal operators
|
||||||
raise CommandArgError(
|
|
||||||
"expected braced word or word without substitutions in argument"
|
|
||||||
" interpreted as list"
|
|
||||||
)
|
|
||||||
|
|
||||||
ts = Lexer(pos=node.contents_pos)
|
if ts.value() in {"*", "&", "|"}:
|
||||||
ts.input(node.contents)
|
# one or two of these characters are legal operators
|
||||||
|
operator = ts.value()
|
||||||
DELIMITERS = {TOK_WS, TOK_BACKSLASH_NEWLINE, TOK_NEWLINE}
|
ts.next()
|
||||||
|
if ts.value() == operator:
|
||||||
list_node = List(pos=node.pos, end_pos=node.end_pos)
|
operator += ts.value()
|
||||||
while ts.type() is not TOK_EOF:
|
|
||||||
while ts.type() in DELIMITERS:
|
|
||||||
ts.next()
|
ts.next()
|
||||||
|
elif ts.value() in {"<", ">"}:
|
||||||
if ts.type() is TOK_EOF:
|
operator = ts.value()
|
||||||
break
|
ts.next()
|
||||||
|
if ts.value() in {operator, "="}:
|
||||||
if ts.type() == TOK_LBRACE:
|
operator += ts.value()
|
||||||
# we can reuse parse_braced_word, since it doesn't use
|
ts.next()
|
||||||
# substitutions in any case
|
elif ts.value() in {"=", "!"}:
|
||||||
list_node.add(self.parse_braced_word(ts))
|
operator = ts.value()
|
||||||
elif ts.type() == TOK_QUOTE:
|
ts.next()
|
||||||
quote_word_pos = ts.pos()
|
if ts.value() != "=":
|
||||||
|
raise TclSyntaxError(
|
||||||
ts.assert_(TOK_QUOTE)
|
f"invalid operator in expression: {operator}", pos, ts.pos()
|
||||||
|
)
|
||||||
bare_word_pos = ts.pos()
|
operator += ts.value()
|
||||||
contents = ""
|
ts.next()
|
||||||
while ts.type() not in {TOK_QUOTE, TOK_EOF}:
|
elif ts.value() in {"*", "/", "%", "+", "-", "^", "eq", "ne", "in", "ni"}:
|
||||||
contents += ts.value()
|
operator = ts.value()
|
||||||
ts.next()
|
ts.next()
|
||||||
word = BareWord(contents, pos=bare_word_pos, end_pos=ts.pos())
|
else:
|
||||||
|
message = "invalid operator in expression: "
|
||||||
ts.expect(
|
if ts.value() == "\\ ":
|
||||||
TOK_QUOTE,
|
message += (
|
||||||
message="reached EOF without finding match for quote",
|
"\\ (check for trailing whitespace if it's the end of the line)"
|
||||||
pos=quote_word_pos,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
list_node.add(QuotedWord(word, pos=quote_word_pos, end_pos=ts.pos()))
|
|
||||||
else:
|
else:
|
||||||
pos = ts.pos()
|
message += ts.value()
|
||||||
contents = ""
|
raise TclSyntaxError(message, pos, ts.pos())
|
||||||
while ts.type() not in {*DELIMITERS, TOK_EOF}:
|
|
||||||
contents += ts.value()
|
|
||||||
ts.next()
|
|
||||||
list_node.add(BareWord(contents, pos=pos, end_pos=ts.pos()))
|
|
||||||
|
|
||||||
return list_node
|
return BareWord(operator, pos=pos, end_pos=ts.pos())
|
||||||
|
|
||||||
def parse_expression(self, node):
|
|
||||||
if node.contents is None:
|
|
||||||
raise CommandArgError(
|
|
||||||
"expected braced word or word without substitutions in argument"
|
|
||||||
" interpreted as expr"
|
|
||||||
)
|
|
||||||
|
|
||||||
ts = Lexer(pos=node.contents_pos)
|
|
||||||
ts.input(node.contents)
|
|
||||||
|
|
||||||
contents = self._parse_expression(ts)
|
|
||||||
ts.expect(
|
|
||||||
TOK_EOF,
|
|
||||||
message=f"expected end of expression, got {ts.value()}",
|
|
||||||
pos=ts.pos(),
|
|
||||||
)
|
|
||||||
if isinstance(node, BracedWord):
|
|
||||||
return BracedExpression(contents, pos=node.pos, end_pos=node.end_pos)
|
|
||||||
|
|
||||||
return Expression(contents, pos=node.pos, end_pos=node.end_pos)
|
|
||||||
|
|||||||
Reference in New Issue
Block a user