diff --git a/Coder.Test/Languages/StringLiteralControlCharacterTests.cs b/Coder.Test/Languages/StringLiteralControlCharacterTests.cs
new file mode 100644
index 0000000..6101016
--- /dev/null
+++ b/Coder.Test/Languages/StringLiteralControlCharacterTests.cs
@@ -0,0 +1,82 @@
+// Copyright (c) 2023-2026 ktsu-dev contributors
+
+namespace ktsu.Coder.Test.Languages;
+
+using System.Linq;
+using ktsu.Coder.Ast;
+using ktsu.Coder.Languages;
+using Microsoft.VisualStudio.TestTools.UnitTesting;
+
+///
+/// Tests that a string literal's control characters and Unicode line terminators are written as escapes.
+///
+///
+/// Written raw, a NUL is a compile error in Go and Python, and U+0085, U+2028 and U+2029 end the line
+/// inside a C# string. The other controls compile but leave invisible bytes in the generated source.
+///
+[TestClass]
+public class StringLiteralControlCharacterTests
+{
+ ///
+ /// NUL, ESC, DEL, a newline, NEL, and the two Unicode line terminators, with a hex digit after ESC
+ /// so a greedy \x escape would swallow it.
+ ///
+ private const string Value = "\0\u001bb\u007f\n\u0085\u2028\u2029";
+
+ ///
+ /// Tests that each generator writes the string's controls as its own escapes, with no raw control
+ /// character left in the output.
+ ///
+ /// The generator's language identifier.
+ /// The literal the generator should write.
+ [TestMethod]
+ [DataRow("csharp", @"""\u0000\u001Bb\u007F\n\u0085\u2028\u2029""")]
+ [DataRow("javascript", @"""\u0000\u001Bb\u007F\n\u0085\u2028\u2029""")]
+ [DataRow("python", @"""\u0000\u001Bb\u007F\n\u0085\u2028\u2029""")]
+ [DataRow("go", @"""\u0000\u001Bb\u007F\n\u0085\u2028\u2029""")]
+ [DataRow("rust", @"""\u{0000}\u{001B}b\u{007F}\n\u{0085}\u{2028}\u{2029}""")]
+ [DataRow("c", @"""\000\033b\177\n\302\205\342\200\250\342\200\251""")]
+ [DataRow("cpp", @"""\000\033b\177\n\302\205\342\200\250\342\200\251""")]
+ public void ControlCharactersAreEscaped(string language, string expected)
+ {
+ ILanguageGenerator generator = Generators().Single(g => g.LanguageId == language);
+
+ string code = generator.Generate(Literal.Text(Value));
+
+ Assert.Contains(expected, code, StringComparison.Ordinal, $"{language} wrote {code}");
+ Assert.IsFalse(code.Any(IsRawControl), $"{language} left a raw control character in {code}");
+ }
+
+ ///
+ /// Tests that a string with nothing to escape numerically is written as before.
+ ///
+ [TestMethod]
+ public void PrintableTextIsWrittenAsIs()
+ {
+ foreach (ILanguageGenerator generator in Generators())
+ {
+ string code = generator.Generate(Literal.Text("caf\u00e9 \"x\"\t"));
+
+ Assert.Contains(@"""café \""x\""\t""", code, StringComparison.Ordinal, $"{generator.LanguageId} wrote {code}");
+ }
+ }
+
+ ///
+ /// Reports whether a character is one the generators must not leave raw inside a literal. A line
+ /// break between statements is expected, so only one inside a literal would be wrong, and the
+ /// expected literal already pins that.
+ ///
+ private static bool IsRawControl(char c) =>
+ c is (< '\u0020' and not '\n' and not '\r') or '\u007f' or '\u0085' or '\u2028' or '\u2029';
+
+ private static ILanguageGenerator[] Generators() =>
+ [
+ new CSharpGenerator(),
+ new CGenerator(),
+ new CppGenerator(),
+ new GoGenerator(),
+ new RustGenerator(),
+ new PythonGenerator(),
+ new JavaScriptGenerator(),
+ ];
+}
diff --git a/Coder/Languages/CFamilyGenerator.cs b/Coder/Languages/CFamilyGenerator.cs
index a603981..8d4ae83 100644
--- a/Coder/Languages/CFamilyGenerator.cs
+++ b/Coder/Languages/CFamilyGenerator.cs
@@ -2,6 +2,9 @@
namespace ktsu.Coder.Languages;
+using System;
+using System.Linq;
+using System.Text;
using ktsu.Coder.Ast;
using ktsu.CodeBlocker;
@@ -166,4 +169,14 @@ protected override void WriteDesignator(string name, CodeBlocker code)
///
protected void WriteBracedList(ConstructionExpression construction, CodeBlocker code, string emptyList) =>
WriteElementList(construction, code, "{", "}", emptyList);
+
+ ///
+ ///
+ /// Octal, one escape per byte of the character's UTF-8 encoding. \x would read on through
+ /// any hex digit that follows it, so ESC followed by b would be the single escape
+ /// \x1bb, and a universal character name may not name a control character. The bytes are
+ /// the ones the character would have been written as raw, so the string's contents do not change.
+ ///
+ protected override string EscapeCodeUnit(char c) =>
+ string.Concat(Encoding.UTF8.GetBytes(c.ToString()).Select(b => $"\\{Convert.ToString(b, 8).PadLeft(3, '0')}"));
}
diff --git a/Coder/Languages/LanguageGeneratorBase.cs b/Coder/Languages/LanguageGeneratorBase.cs
index 38ac6d7..31bd09b 100644
--- a/Coder/Languages/LanguageGeneratorBase.cs
+++ b/Coder/Languages/LanguageGeneratorBase.cs
@@ -6,6 +6,7 @@ namespace ktsu.Coder.Languages;
using System.Collections.Generic;
using System.Globalization;
using System.Linq;
+using System.Text;
using ktsu.Coder.Ast;
using ktsu.CodeBlocker;
@@ -792,20 +793,58 @@ protected static Visibility VisibilityOf(AstNode node) =>
///
/// The raw string value.
/// The escaped value, without surrounding quotes.
- /// Every language the generators target uses these escapes, Python included.
- protected static string EscapeString(string value)
+ ///
+ /// Every language the generators target uses the named escapes, Python included. Every other
+ /// control character, and the Unicode line terminators, are escaped numerically through
+ /// , because written raw they break the literal: Go and Python reject a
+ /// NUL in source, and C# reads U+0085, U+2028 and U+2029 as ending the line.
+ ///
+ protected string EscapeString(string value)
{
Ensure.NotNull(value);
// Ordinal explicitly: these are source-syntax escapes, never subject to a culture.
- return value
+ string named = value
.Replace("\\", "\\\\", StringComparison.Ordinal)
.Replace("\"", "\\\"", StringComparison.Ordinal)
.Replace("\n", "\\n", StringComparison.Ordinal)
.Replace("\r", "\\r", StringComparison.Ordinal)
.Replace("\t", "\\t", StringComparison.Ordinal);
+
+ if (!named.Any(NeedsNumericEscape))
+ {
+ return named;
+ }
+
+ StringBuilder escaped = new(named.Length + 8);
+ foreach (char c in named)
+ {
+ escaped.Append(NeedsNumericEscape(c) ? EscapeCodeUnit(c) : c.ToString());
+ }
+
+ return escaped.ToString();
}
+ ///
+ /// Spells one character that cannot stand raw in a string literal as a numeric escape.
+ ///
+ /// A control character or Unicode line terminator.
+ /// The escape sequence.
+ ///
+ /// Defaults to \uXXXX, which C#, JavaScript, Python and Go all read. Rust writes the code
+ /// point in braces, and C and C++ override this with octal, since their \x is greedy and
+ /// would swallow a following hex digit.
+ ///
+ protected virtual string EscapeCodeUnit(char c) => $"\\u{(int)c:X4}";
+
+ ///
+ /// Reports whether a character must be written as a numeric escape inside a string literal.
+ ///
+ /// The character.
+ /// True for a C0 control, DEL, U+0085, U+2028 or U+2029.
+ private static bool NeedsNumericEscape(char c) =>
+ c is < '\u0020' or '\u007F' or '\u0085' or '\u2028' or '\u2029';
+
///
/// Maps a binary operator to its C-family spelling.
///
diff --git a/Coder/Languages/RustGenerator.cs b/Coder/Languages/RustGenerator.cs
index c0b4488..40165aa 100644
--- a/Coder/Languages/RustGenerator.cs
+++ b/Coder/Languages/RustGenerator.cs
@@ -1649,4 +1649,8 @@ private static string Borrow(string owned)
? $"[{owned[4..^1]}]"
: owned;
}
+
+ ///
+ /// Rust writes a Unicode escape with the code point in braces.
+ protected override string EscapeCodeUnit(char c) => $"\\u{{{(int)c:X4}}}";
}