diff --git a/Coder.Test/Languages/StringLiteralControlCharacterTests.cs b/Coder.Test/Languages/StringLiteralControlCharacterTests.cs new file mode 100644 index 0000000..6101016 --- /dev/null +++ b/Coder.Test/Languages/StringLiteralControlCharacterTests.cs @@ -0,0 +1,82 @@ +// Copyright (c) 2023-2026 ktsu-dev contributors + +namespace ktsu.Coder.Test.Languages; + +using System.Linq; +using ktsu.Coder.Ast; +using ktsu.Coder.Languages; +using Microsoft.VisualStudio.TestTools.UnitTesting; + +/// +/// Tests that a string literal's control characters and Unicode line terminators are written as escapes. +/// +/// +/// Written raw, a NUL is a compile error in Go and Python, and U+0085, U+2028 and U+2029 end the line +/// inside a C# string. The other controls compile but leave invisible bytes in the generated source. +/// +[TestClass] +public class StringLiteralControlCharacterTests +{ + /// + /// NUL, ESC, DEL, a newline, NEL, and the two Unicode line terminators, with a hex digit after ESC + /// so a greedy \x escape would swallow it. + /// + private const string Value = "\0\u001bb\u007f\n\u0085\u2028\u2029"; + + /// + /// Tests that each generator writes the string's controls as its own escapes, with no raw control + /// character left in the output. + /// + /// The generator's language identifier. + /// The literal the generator should write. + [TestMethod] + [DataRow("csharp", @"""\u0000\u001Bb\u007F\n\u0085\u2028\u2029""")] + [DataRow("javascript", @"""\u0000\u001Bb\u007F\n\u0085\u2028\u2029""")] + [DataRow("python", @"""\u0000\u001Bb\u007F\n\u0085\u2028\u2029""")] + [DataRow("go", @"""\u0000\u001Bb\u007F\n\u0085\u2028\u2029""")] + [DataRow("rust", @"""\u{0000}\u{001B}b\u{007F}\n\u{0085}\u{2028}\u{2029}""")] + [DataRow("c", @"""\000\033b\177\n\302\205\342\200\250\342\200\251""")] + [DataRow("cpp", @"""\000\033b\177\n\302\205\342\200\250\342\200\251""")] + public void ControlCharactersAreEscaped(string language, string expected) + { + ILanguageGenerator generator = Generators().Single(g => g.LanguageId == language); + + string code = generator.Generate(Literal.Text(Value)); + + Assert.Contains(expected, code, StringComparison.Ordinal, $"{language} wrote {code}"); + Assert.IsFalse(code.Any(IsRawControl), $"{language} left a raw control character in {code}"); + } + + /// + /// Tests that a string with nothing to escape numerically is written as before. + /// + [TestMethod] + public void PrintableTextIsWrittenAsIs() + { + foreach (ILanguageGenerator generator in Generators()) + { + string code = generator.Generate(Literal.Text("caf\u00e9 \"x\"\t")); + + Assert.Contains(@"""café \""x\""\t""", code, StringComparison.Ordinal, $"{generator.LanguageId} wrote {code}"); + } + } + + /// + /// Reports whether a character is one the generators must not leave raw inside a literal. A line + /// break between statements is expected, so only one inside a literal would be wrong, and the + /// expected literal already pins that. + /// + private static bool IsRawControl(char c) => + c is (< '\u0020' and not '\n' and not '\r') or '\u007f' or '\u0085' or '\u2028' or '\u2029'; + + private static ILanguageGenerator[] Generators() => + [ + new CSharpGenerator(), + new CGenerator(), + new CppGenerator(), + new GoGenerator(), + new RustGenerator(), + new PythonGenerator(), + new JavaScriptGenerator(), + ]; +} diff --git a/Coder/Languages/CFamilyGenerator.cs b/Coder/Languages/CFamilyGenerator.cs index a603981..8d4ae83 100644 --- a/Coder/Languages/CFamilyGenerator.cs +++ b/Coder/Languages/CFamilyGenerator.cs @@ -2,6 +2,9 @@ namespace ktsu.Coder.Languages; +using System; +using System.Linq; +using System.Text; using ktsu.Coder.Ast; using ktsu.CodeBlocker; @@ -166,4 +169,14 @@ protected override void WriteDesignator(string name, CodeBlocker code) /// protected void WriteBracedList(ConstructionExpression construction, CodeBlocker code, string emptyList) => WriteElementList(construction, code, "{", "}", emptyList); + + /// + /// + /// Octal, one escape per byte of the character's UTF-8 encoding. \x would read on through + /// any hex digit that follows it, so ESC followed by b would be the single escape + /// \x1bb, and a universal character name may not name a control character. The bytes are + /// the ones the character would have been written as raw, so the string's contents do not change. + /// + protected override string EscapeCodeUnit(char c) => + string.Concat(Encoding.UTF8.GetBytes(c.ToString()).Select(b => $"\\{Convert.ToString(b, 8).PadLeft(3, '0')}")); } diff --git a/Coder/Languages/LanguageGeneratorBase.cs b/Coder/Languages/LanguageGeneratorBase.cs index 38ac6d7..31bd09b 100644 --- a/Coder/Languages/LanguageGeneratorBase.cs +++ b/Coder/Languages/LanguageGeneratorBase.cs @@ -6,6 +6,7 @@ namespace ktsu.Coder.Languages; using System.Collections.Generic; using System.Globalization; using System.Linq; +using System.Text; using ktsu.Coder.Ast; using ktsu.CodeBlocker; @@ -792,20 +793,58 @@ protected static Visibility VisibilityOf(AstNode node) => /// /// The raw string value. /// The escaped value, without surrounding quotes. - /// Every language the generators target uses these escapes, Python included. - protected static string EscapeString(string value) + /// + /// Every language the generators target uses the named escapes, Python included. Every other + /// control character, and the Unicode line terminators, are escaped numerically through + /// , because written raw they break the literal: Go and Python reject a + /// NUL in source, and C# reads U+0085, U+2028 and U+2029 as ending the line. + /// + protected string EscapeString(string value) { Ensure.NotNull(value); // Ordinal explicitly: these are source-syntax escapes, never subject to a culture. - return value + string named = value .Replace("\\", "\\\\", StringComparison.Ordinal) .Replace("\"", "\\\"", StringComparison.Ordinal) .Replace("\n", "\\n", StringComparison.Ordinal) .Replace("\r", "\\r", StringComparison.Ordinal) .Replace("\t", "\\t", StringComparison.Ordinal); + + if (!named.Any(NeedsNumericEscape)) + { + return named; + } + + StringBuilder escaped = new(named.Length + 8); + foreach (char c in named) + { + escaped.Append(NeedsNumericEscape(c) ? EscapeCodeUnit(c) : c.ToString()); + } + + return escaped.ToString(); } + /// + /// Spells one character that cannot stand raw in a string literal as a numeric escape. + /// + /// A control character or Unicode line terminator. + /// The escape sequence. + /// + /// Defaults to \uXXXX, which C#, JavaScript, Python and Go all read. Rust writes the code + /// point in braces, and C and C++ override this with octal, since their \x is greedy and + /// would swallow a following hex digit. + /// + protected virtual string EscapeCodeUnit(char c) => $"\\u{(int)c:X4}"; + + /// + /// Reports whether a character must be written as a numeric escape inside a string literal. + /// + /// The character. + /// True for a C0 control, DEL, U+0085, U+2028 or U+2029. + private static bool NeedsNumericEscape(char c) => + c is < '\u0020' or '\u007F' or '\u0085' or '\u2028' or '\u2029'; + /// /// Maps a binary operator to its C-family spelling. /// diff --git a/Coder/Languages/RustGenerator.cs b/Coder/Languages/RustGenerator.cs index c0b4488..40165aa 100644 --- a/Coder/Languages/RustGenerator.cs +++ b/Coder/Languages/RustGenerator.cs @@ -1649,4 +1649,8 @@ private static string Borrow(string owned) ? $"[{owned[4..^1]}]" : owned; } + + /// + /// Rust writes a Unicode escape with the code point in braces. + protected override string EscapeCodeUnit(char c) => $"\\u{{{(int)c:X4}}}"; }