From e2fa244d4d04e933c875890b14cfee61764a7685 Mon Sep 17 00:00:00 2001 From: matt-edmondson Date: Sun, 27 Sep 2026 07:23:45 +0000 Subject: [PATCH] [patch] Recognise an opening delimiter with trailing whitespace or a leading BOM HasFrontmatter required the document to start with exactly "---" and a newline, while closing delimiters already allowed trailing whitespace. A header opened by "--- " or "---\t", or preceded by a byte order mark, was therefore invisible, and AddFrontmatter stacked a second header above it. The opening line now goes through IsDelimiterLine like every other delimiter, after skipping a leading BOM, both in HasFrontmatter and in the block splitter. Fixes #141 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01KMqfUfzWj2Gmdz1EqYdnLe --- Frontmatter.Test/CombineFrontmatterTests.cs | 2 +- Frontmatter.Test/DelimiterLineTests.cs | 48 +++++++++++++++++++++ Frontmatter/Frontmatter.cs | 48 ++++++++------------- 3 files changed, 68 insertions(+), 30 deletions(-) diff --git a/Frontmatter.Test/CombineFrontmatterTests.cs b/Frontmatter.Test/CombineFrontmatterTests.cs index 3624f5c..d34f2f6 100644 --- a/Frontmatter.Test/CombineFrontmatterTests.cs +++ b/Frontmatter.Test/CombineFrontmatterTests.cs @@ -253,7 +253,7 @@ public void CombineFrontmatter_WithComplexPropertyValues_PreservesStructure() public void CombineFrontmatter_WithHashCollidingDocuments_ProcessesEachIndependently() { // Arrange - "\n" rather than Environment.NewLine so the bytes hashed are the same on every - // platform; DetectNewLine reads the document's own convention, so both still parse. + // platform; parsing follows the document's own line endings, so both still parse. string first = "---\ntitle: Release Notes 281277\n---\n"; string second = "---\ntitle: Release Notes 1084130\n---\n"; Assert.AreNotEqual(first, second, "The two documents must be distinct for this test to mean anything"); diff --git a/Frontmatter.Test/DelimiterLineTests.cs b/Frontmatter.Test/DelimiterLineTests.cs index 4ae1ec7..f418972 100644 --- a/Frontmatter.Test/DelimiterLineTests.cs +++ b/Frontmatter.Test/DelimiterLineTests.cs @@ -106,4 +106,52 @@ public void AddFrontmatter_ExistingFrontmatterIsEmpty_AddsTheProperties() Assert.AreEqual($"---{Nl}author: B{Nl}---{Nl}Body{Nl}", result); } + + [TestMethod] + [DataRow("--- ", DisplayName = "Trailing space")] + [DataRow("---\t", DisplayName = "Trailing tab")] + [DataRow("\uFEFF---", DisplayName = "Byte order mark")] + public void HasFrontmatter_OpeningDelimiterHasTrailingWhitespaceOrBom_RecognisesTheHeader(string opening) + { + string input = $"{opening}{Nl}title: A{Nl}---{Nl}Body{Nl}"; + + Assert.IsTrue(Frontmatter.HasFrontmatter(input)); + } + + [TestMethod] + [DataRow("--- ", DisplayName = "Trailing space")] + [DataRow("---\t", DisplayName = "Trailing tab")] + [DataRow("\uFEFF---", DisplayName = "Byte order mark")] + public void ExtractFrontmatter_OpeningDelimiterHasTrailingWhitespaceOrBom_ReadsTheHeader(string opening) + { + string input = $"{opening}{Nl}title: A{Nl}---{Nl}Body{Nl}"; + + Dictionary? frontmatter = Frontmatter.ExtractFrontmatter(input); + + Assert.IsNotNull(frontmatter); + Assert.HasCount(1, frontmatter); + Assert.AreEqual("A", frontmatter["title"]); + Assert.AreEqual("Body", Frontmatter.ExtractBody(input)); + } + + [TestMethod] + [DataRow("--- ", DisplayName = "Trailing space")] + [DataRow("---\t", DisplayName = "Trailing tab")] + [DataRow("\uFEFF---", DisplayName = "Byte order mark")] + public void AddFrontmatter_OpeningDelimiterHasTrailingWhitespaceOrBom_MergesIntoTheExistingHeader(string opening) + { + string input = $"{opening}{Nl}title: A{Nl}---{Nl}Body{Nl}"; + + string result = Frontmatter.AddFrontmatter(input, new Dictionary { ["author"] = "B" }); + + Assert.AreEqual($"---{Nl}title: A{Nl}author: B{Nl}---{Nl}Body{Nl}", result); + } + + [TestMethod] + public void HasFrontmatter_DelimiterWithoutLineEnding_IsNotFrontmatter() + { + Assert.IsFalse(Frontmatter.HasFrontmatter("--- ")); + Assert.IsFalse(Frontmatter.HasFrontmatter("\uFEFF")); + Assert.IsFalse(Frontmatter.HasFrontmatter(string.Empty)); + } } diff --git a/Frontmatter/Frontmatter.cs b/Frontmatter/Frontmatter.cs index 85bc8ed..18a4a8f 100644 --- a/Frontmatter/Frontmatter.cs +++ b/Frontmatter/Frontmatter.cs @@ -6,8 +6,6 @@ namespace ktsu.Frontmatter; using System.Collections.Concurrent; using System.Collections.Generic; -using ktsu.Extensions; - /// /// Provides methods for processing and manipulating YAML frontmatter in markdown files. /// @@ -18,6 +16,11 @@ public static class Frontmatter /// private const string FrontmatterDelimiter = "---"; + /// + /// The byte order mark a document may begin with when it was decoded without stripping it. + /// + private const char ByteOrderMark = '\uFEFF'; + /// /// Cache for processed frontmatter to avoid repeated processing of identical content. /// Keyed by the document text together with the option flags it was processed under, so a cache @@ -148,9 +151,21 @@ public static bool HasFrontmatter(string input) { Ensure.NotNull(input); - return !string.IsNullOrEmpty(input) && input.StartsWithOrdinal(FrontmatterDelimiter + DetectNewLine(input)); + // The opening delimiter follows the same rule as the closing one, so trailing whitespace an editor + // left behind does not hide the header. A leading byte order mark is skipped, since callers that + // decode bytes or streams themselves keep it where File.ReadAllText would strip it. + int start = OpeningLineStart(input); + int end = input.IndexOfAny(['\r', '\n'], start); + return end >= 0 && IsDelimiterLine(input[start..end]); } + /// + /// Finds where the first line of a document starts, skipping a leading byte order mark. + /// + /// The document to inspect. + /// The offset of the first character after any byte order mark. + private static int OpeningLineStart(string input) => input.Length > 0 && input[0] == ByteOrderMark ? 1 : 0; + /// /// Adds frontmatter to a markdown document that doesn't already have it. /// @@ -282,32 +297,6 @@ private static Dictionary SortFrontmatterProperties(Dictionary - /// Finds the line ending a document actually uses, rather than assuming the host's. - /// - /// - /// A markdown file written on Windows still holds CRLF when it is read on Linux, and one - /// written on Linux still holds LF when it is read on Windows. Line endings travel with the - /// document, so parsing has to follow the document rather than the machine it is parsed on. - /// Writing is a separate question and still uses the host convention. - /// - /// The document to inspect. - /// The ending that terminates the first line, or the host's when there is none. - private static string DetectNewLine(string input) - { - // The first line is the opening delimiter, so its terminator is the one the delimiter - // search has to match. Scanning the whole document instead would pick up a stray ending - // from the body, and a document written here with LF but quoting CRLF content would then - // report CRLF and fail to match the delimiter it had just written itself. - int index = input.IndexOfAny(['\r', '\n']); - - return index < 0 - ? Environment.NewLine - : input[index] == '\n' ? "\n" - : index + 1 < input.Length && input[index + 1] == '\n' ? "\r\n" - : "\r"; - } - /// /// Extracts all frontmatter objects from a markdown document. /// @@ -366,6 +355,7 @@ private static bool TrySplitFrontmatterBlocks(string input, out List blo } List<(int Start, int End)> lines = SplitLines(input); + lines[0] = (OpeningLineStart(input), lines[0].End); bool IsDelimiterAt(int index) => IsDelimiterLine(input[lines[index].Start..lines[index].End]); int next = 0;