add some features

This commit is contained in:
2025-08-03 21:04:14 +02:00
parent 5e2e85830b
commit 0318532308
8 changed files with 374 additions and 305 deletions
+336 -123
View File
@@ -1,134 +1,347 @@
from tclint.parser import Parser, _strip_ws
from tclint.lexer import (
STATE_BRACEDWORD,
TOK_BACKSLASH_NEWLINE,
TOK_EOF,
TOK_LBRACE,
TOK_RBRACE,
TOK_WS,
TOK_NEWLINE,
TOK_ALPHA_CHARS,
TOK_RPAREN,
TOK_LPAREN,
TOK_DOLLAR,
TOK_QUOTE,
TOK_LBRACKET,
TOK_NUM_CHARS,
TclSyntaxError,
)
from tclint.parser import Parser
from tclint.commands import CommandArgError
from tclint.syntax_tree import (
BracedWord,
BareWord,
BinaryOp,
TernaryOp,
ParenExpression,
UnaryOp,
BracedExpression,
List,
QuotedWord,
Expression,
)
import ply.lex as lex
from typing import Tuple
TOK_BACKSLASH_NEWLINE = "BACKSLASH_NEWLINE"
TOK_BACKSLASH_SUB = "BACKSLASH_SUB"
TOK_NEWLINE = "NEWLINE"
TOK_SEMI = "SEMI"
TOK_WS = "WS"
TOK_QUOTE = "QUOTE"
TOK_ARG_EXPANSION = "ARG_EXPANSION"
TOK_LBRACE = "LBRACE"
TOK_RBRACE = "RBRACE"
TOK_STAR = "STAR"
TOK_LBRACKET = "LBRACKET"
TOK_RBRACKET = "RBRACKET"
TOK_DOLLAR = "DOLLAR"
TOK_LPAREN = "LPAREN"
TOK_RPAREN = "RPAREN"
TOK_HASH = "HASH"
TOK_ALPHA_CHARS = "ALPHA_CHARS"
TOK_NUM_CHARS = "NUM_CHARS"
TOK_NAMESPACE_SEP = "NAMESPACE_SEP"
TOK_CHAR = "CHAR"
TOK_CONTENTS = "CONTENTS"
TOK_EOF = None
STATE_BRACEDWORD = "bracedword"
class TclSyntaxError(Exception):
def __init__(self, message, start: Tuple[int, int], end: Tuple[int, int]):
super().__init__(message)
self.start = start
self.end = end
class _LexTable:
tokens = (
TOK_BACKSLASH_NEWLINE,
TOK_BACKSLASH_SUB,
TOK_NEWLINE,
TOK_SEMI,
TOK_WS,
TOK_QUOTE,
TOK_ARG_EXPANSION,
TOK_LBRACE,
TOK_RBRACE,
TOK_STAR,
TOK_LBRACKET,
TOK_RBRACKET,
TOK_DOLLAR,
TOK_LPAREN,
TOK_RPAREN,
TOK_HASH,
TOK_ALPHA_CHARS,
TOK_NUM_CHARS,
TOK_NAMESPACE_SEP,
TOK_CHAR,
TOK_CONTENTS,
)
# This defines a conditional lexing state for parsing braced words. This is a
# performance optimization; since there are few special characters in this context,
# we can use a smaller set of tokens to parse them faster. This has a large impact
# since most Tcl programs have a large number of braced words. Any token with
# `bracedword` in its name is included in this state. Tokens that are included in
# this state and the default state also include `INITIAL` in their name.
states = ((STATE_BRACEDWORD, "exclusive"),)
def _tok(self, t):
pos = (t.lexer.lineno, t.lexer.colno)
t.lexer.lineno += t.value.count("\n")
index = t.value.rfind("\n")
if index == -1:
t.lexer.colno += len(t.value)
else:
remaining = t.value[index + 1 :]
t.lexer.colno = len(remaining) + 1
t.value = (t.value, pos)
return t
# Priority important
def t_bracedword_INITIAL_BACKSLASH_NEWLINE(self, t):
r"\\\r?\n"
return self._tok(t)
# Priority important
def t_bracedword_INITIAL_BACKSLASH_SUB(self, t):
r"\\."
return self._tok(t)
def t_NEWLINE(self, t):
r"\n"
return self._tok(t)
def t_SEMI(self, t):
r";"
return self._tok(t)
# TODO: should use \s?
def t_WS(self, t):
r"[\t\v\f\r ]+"
return self._tok(t)
def t_QUOTE(self, t):
r'"'
return self._tok(t)
# Must be higher priority than LBRACE
def t_ARG_EXPANSION(self, t):
r"\{\*\}"
return self._tok(t)
def t_bracedword_INITIAL_LBRACE(self, t):
r"\{"
return self._tok(t)
def t_bracedword_INITIAL_RBRACE(self, t):
r"\}"
return self._tok(t)
def t_STAR(self, t):
r"\*"
return self._tok(t)
def t_LBRACKET(self, t):
r"\["
return self._tok(t)
def t_RBRACKET(self, t):
r"\]"
return self._tok(t)
def t_DOLLAR(self, t):
r"\$"
return self._tok(t)
def t_LPAREN(self, t):
r"\("
return self._tok(t)
def t_RPAREN(self, t):
r"\)"
return self._tok(t)
def t_HASH(self, t):
r"\#"
return self._tok(t)
# Valid non-numeric chars in variable names
def t_ALPHA_CHARS(self, t):
r"[A-Za-z_]+"
return self._tok(t)
# Valid numeric chars in variable names
# This is split up from the above to facilitate expression parsing, since
# e.g. 1eq1 can't be a single token.
def t_NUM_CHARS(self, t):
r"[0-9]+"
return self._tok(t)
def t_NAMESPACE_SEP(self, t):
r"::+"
return self._tok(t)
def t_bracedword_CONTENTS(self, t):
r"[^{}\\]+"
return self._tok(t)
# Catch-all. TODO: inefficient, should probably munch multiple chars
def t_CHAR(self, t):
r"."
return self._tok(t)
# Error handling rule
# TODO: do we need this? since we have a catch-all...
# there is a warning
def t_bracedword_INITIAL_error(self, t):
print("Illegal character '%s'" % t.value[0])
t.lexer.skip(1)
def __init__(self):
self.lexer = lex.lex(object=self)
self.lexer.lineno = 1
self.lexer.colno = 1
def new_lexer(self, pos=None):
lexer = self.lexer.clone()
lexer.lineno = 1
lexer.colno = 1
if pos is not None:
line, col = pos
lexer.lineno = line
lexer.colno = col
return lexer
# Calling `lex.lex()` performs an expensive reflection process to generate the lexer.
# This singleton class holds a preinitialized lexer that can then be cloned to create
# individual instances.
LexTable = _LexTable()
class Lexer:
def __init__(self, pos=None):
self.lexer = LexTable.new_lexer(pos)
self.current = None
def input(self, text):
self.lexer.input(text)
self.current = self.lexer.token()
def type(self):
if self.current is None:
return TOK_EOF
return self.current.type
def value(self):
if self.current is None:
return None
return self.current.value[0]
def pos(self):
if self.current is None:
return (self.lexer.lineno, self.lexer.colno)
return self.current.value[1]
def next(self):
self.current = self.lexer.token()
def expect(self, *tokens, message, pos):
if self.type() not in tokens:
self.next() # munch another token to update position
raise TclSyntaxError(message, pos, self.pos())
self.next()
def assert_(self, *tokens):
assert self.current.type in tokens
self.next()
class CustomParser(Parser):
@_strip_ws
def _parse_expression(self, ts):
op1 = self._parse_operand(ts)
expr = op1
def parse(self, script, pos=None):
lexer = Lexer(pos=pos)
lexer.input(script)
tree = self._parse_script(lexer, in_command_sub=False)
assert lexer.type() == TOK_EOF, (
"Didn't reach EOF parsing script, please file a bug report."
)
# Add TOK_BACKSLASH_NEWLINE to the tokens we need to skip
while ts.type() == TOK_BACKSLASH_NEWLINE:
ts.next()
return tree
# last condition is hack to break out of expression in case we're in ternary op
if ts.type() not in {
def parse_list(self, node):
"""Parse contents of node as Tcl list. This is a distinct entry point
that doesn't get used when generating the main syntax tree, but is used
in command-specific argument parsing.
"""
if isinstance(node, List):
return node
if node.contents is None:
raise CommandArgError(
"expected braced word or word without substitutions in argument"
" interpreted as list"
)
ts = Lexer(pos=node.contents_pos)
ts.input(node.contents)
DELIMITERS = {TOK_WS, TOK_BACKSLASH_NEWLINE, TOK_NEWLINE}
list_node = List(pos=node.pos, end_pos=node.end_pos)
while ts.type() is not TOK_EOF:
while ts.type() in DELIMITERS:
ts.next()
if ts.type() is TOK_EOF:
break
if ts.type() == TOK_LBRACE:
# we can reuse parse_braced_word, since it doesn't use
# substitutions in any case
list_node.add(self.parse_braced_word(ts))
elif ts.type() == TOK_QUOTE:
quote_word_pos = ts.pos()
ts.assert_(TOK_QUOTE)
bare_word_pos = ts.pos()
contents = ""
while ts.type() not in {TOK_QUOTE, TOK_EOF}:
contents += ts.value()
ts.next()
word = BareWord(contents, pos=bare_word_pos, end_pos=ts.pos())
ts.expect(
TOK_QUOTE,
message="reached EOF without finding match for quote",
pos=quote_word_pos,
)
list_node.add(QuotedWord(word, pos=quote_word_pos, end_pos=ts.pos()))
else:
pos = ts.pos()
contents = ""
while ts.type() not in {*DELIMITERS, TOK_EOF}:
contents += ts.value()
ts.next()
list_node.add(BareWord(contents, pos=pos, end_pos=ts.pos()))
return list_node
def parse_expression(self, node):
if node.contents is None:
raise CommandArgError(
"expected braced word or word without substitutions in argument"
" interpreted as expr"
)
ts = Lexer(pos=node.contents_pos)
ts.input(node.contents)
contents = self._parse_expression(ts)
ts.expect(
TOK_EOF,
TOK_RPAREN,
TOK_BACKSLASH_NEWLINE,
} and ts.value() not in {":", ","}:
if ts.value() == "?":
# weird hack to record operator
start = ts.pos()
ts.next()
q = BareWord("?", pos=start, end_pos=ts.pos())
message=f"expected end of expression, got {ts.value()}",
pos=ts.pos(),
)
if isinstance(node, BracedWord):
return BracedExpression(contents, pos=node.pos, end_pos=node.end_pos)
op2 = self._parse_expression(ts)
if ts.value() != ":":
start = ts.pos()
ts.next()
end = ts.pos()
raise TclSyntaxError(
"expected ':' to continue ternary expression", start, end
)
# weird hack again
start = ts.pos()
ts.next()
colon = BareWord(":", pos=start, end_pos=ts.pos())
op3 = self._parse_expression(ts)
expr = TernaryOp(
op1, q, op2, colon, op3, pos=op1.pos, end_pos=op3.end_pos
)
else:
operator = self._parse_operator(ts)
while ts.type() == TOK_BACKSLASH_NEWLINE:
ts.next()
op2 = self._parse_expression(ts)
expr = BinaryOp(op1, operator, op2, pos=op1.pos, end_pos=op2.end_pos)
if ts.type() not in (
TOK_RPAREN,
TOK_BACKSLASH_NEWLINE,
) and ts.value() not in {":", ","}:
ts.expect(TOK_EOF, message="expected end of expression", pos=ts.pos())
return expr
def _parse_operator(self, ts):
pos = ts.pos()
# hacky logic to handle parsing legal operators
# Skip any backslash-newlines before the operator
while ts.type() in {TOK_WS, TOK_BACKSLASH_NEWLINE, TOK_NEWLINE}:
ts.next()
if ts.value() in {"&&", "and"}: # Add explicit handling for logical AND
operator = ts.value()
ts.next()
return BareWord(operator, pos=pos, end_pos=ts.pos())
elif ts.value() in {"*", "&", "|"}:
# one or two of these characters are legal operators
operator = ts.value()
ts.next()
if ts.value() == operator:
operator += ts.value()
ts.next()
elif ts.value() in {"<", ">"}:
operator = ts.value()
ts.next()
if ts.value() in {operator, "="}:
operator += ts.value()
ts.next()
elif ts.value() in {"=", "!"}:
operator = ts.value()
ts.next()
if ts.value() != "=":
raise TclSyntaxError(
f"invalid operator in expression: {operator}", pos, ts.pos()
)
operator += ts.value()
ts.next()
elif ts.value() in {"*", "/", "%", "+", "-", "^", "eq", "ne", "in", "ni"}:
operator = ts.value()
ts.next()
else:
while ts.type() in {TOK_WS, TOK_BACKSLASH_NEWLINE, TOK_NEWLINE}:
ts.next()
if ts.value() in {"&&", "and"}: # Try again after whitespace
operator = ts.value()
ts.next()
else:
raise TclSyntaxError(
f"invalid operator in expression: {ts.value()}", pos, ts.pos()
)
return BareWord(operator, pos=pos, end_pos=ts.pos())
return Expression(contents, pos=node.pos, end_pos=node.end_pos)