diff --git a/CHANGELOG b/CHANGELOG index 44d5938e..553ab1d2 100644 --- a/CHANGELOG +++ b/CHANGELOG @@ -1,7 +1,12 @@ Development Version ------------------- -Nothing yet. +* Fixed the lexer mis-tokenizing string literals that contain an escaped + backslash. In ``SELECT '\\', '\\'`` the closing quote was swallowed by the + escape sequence, so the literal ran on and the remainder was emitted as + error tokens. The same ambiguity affected double quoted strings, which are + reported as ``String.Symbol``. See + https://github.com/andialbrecht/sqlparse/issues/814 Release 0.6.0 (Aug 13, 2026) diff --git a/sqlparse/keywords.py b/sqlparse/keywords.py index dd6e5d15..e58fa87a 100644 --- a/sqlparse/keywords.py +++ b/sqlparse/keywords.py @@ -163,9 +163,13 @@ def find_delimited_spans(text): (r'(?![_A-ZÀ-Ü])-?(\d+(\.\d*)|\.\d+)(?![_A-ZÀ-Ü])', tokens.Number.Float), (r'(?![_A-ZÀ-Ü])-?\d+(?![_A-ZÀ-Ü])', tokens.Number.Integer), - (r"'(''|\\'|[^'])*'", tokens.String.Single), + # An escape is either a doubled quote or a backslash followed by any + # character. Using "\\'" for the latter is ambiguous: once "[^']" has + # consumed the first backslash of "\\", the second one pairs up with + # the closing quote and swallows it (issue #814). + (r"'(''|\\.|[^'\\])*'", tokens.String.Single), # not a real string literal in ANSI SQL: - (r'"(""|\\"|[^"])*"', tokens.String.Symbol), + (r'"(""|\\.|[^"\\])*"', tokens.String.Symbol), (r'(""|".*?[^\\]")', tokens.String.Symbol), # sqlite names can be escaped with [square brackets]. left bracket # cannot be preceded by word character or a right bracket -- diff --git a/tests/test_regressions.py b/tests/test_regressions.py index aca7f7b3..c8de22f6 100644 --- a/tests/test_regressions.py +++ b/tests/test_regressions.py @@ -516,3 +516,36 @@ def limit_recursion(): def test_max_recursion(limit_recursion): with pytest.raises(SQLParseError): sqlparse.parse('[' * 1000 + ']' * 1000) + + +def test_issue814(): + # An escaped backslash must not pair up with the closing quote: that + # used to swallow it, leaving the rest of the statement as error tokens. + stmt = sqlparse.parse("SELECT '\\\\', '\\\\'")[0] + assert [t.value for t in stmt.flatten()] == [ + 'SELECT', ' ', "'\\\\'", ',', ' ', "'\\\\'" + ] + + +def test_issue814_quoted_identifier(): + # The double quoted rule had the same ambiguity, it just reports + # String.Symbol rather than String.Single. + stmt = sqlparse.parse('SELECT "\\\\", "\\\\"')[0] + assert [t.value for t in stmt.flatten()] == [ + 'SELECT', ' ', '"\\\\"', ',', ' ', '"\\\\"' + ] + + +@pytest.mark.parametrize("value", ["'it\\'s'", "'it''s'", "'a\\\\'", "'plain'"]) +def test_issue814_keeps_escaping_forms(value): + # The escaping forms that already lexed as a single literal must keep + # doing so. + stmt = sqlparse.parse(f'SELECT {value}')[0] + literals = [t for t in stmt.flatten() if t.ttype is T.Literal.String.Single] + assert [t.value for t in literals] == [value] + + +def test_issue814_unterminated_string_still_unmatched(): + # A literal without a closing quote must keep failing to lex as one. + stmt = sqlparse.parse("SELECT 'abc")[0] + assert [t.value for t in stmt.flatten()] == ['SELECT', ' ', "'", 'abc']