From 2b3bad8ad2db41a65d41e5786df7c65b6a4aa035 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 12:32:21 +0000 Subject: [PATCH] Recognise frontmatter delimiters only as whole lines at the top of a document The document was split on every "---" followed by a newline, anywhere in the text. This caused four problems: - A markdown horizontal rule pair around a "key: value" paragraph was merged into the header. - A value such as "foo---" split the header mid-line. - CombineFrontmatter merged the second block into the header and also left it in the body. - A closing "---" at the very end of the file was not recognised, so AddFrontmatter then overwrote the properties it could not see. Frontmatter is now parsed line by line. The first line opens a block, and the block closes at the next line that is exactly "---" (trailing whitespace allowed), including one at the end of the input. Further blocks are taken only while they follow directly. Every CRLF, LF and CR ending is recognised, so mixed-ending documents still parse. AddFrontmatter no longer treats frontmatter it cannot read as empty. It returns the document unchanged, so a parsing gap can't lose properties. Fixes ktsu-dev/Frontmatter#132 Fixes ktsu-dev/Frontmatter#133 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01VpbpayL5Kw99JfbzgkJfKa --- Frontmatter.Test/DelimiterLineTests.cs | 109 +++++++++++++++++++++++ Frontmatter/Frontmatter.cs | 114 +++++++++++++++++++++++-- 2 files changed, 215 insertions(+), 8 deletions(-) create mode 100644 Frontmatter.Test/DelimiterLineTests.cs diff --git a/Frontmatter.Test/DelimiterLineTests.cs b/Frontmatter.Test/DelimiterLineTests.cs new file mode 100644 index 0000000..4ae1ec7 --- /dev/null +++ b/Frontmatter.Test/DelimiterLineTests.cs @@ -0,0 +1,109 @@ +// Copyright (c) 2023-2026 ktsu-dev contributors + +namespace ktsu.Frontmatter.Test; + +using Microsoft.VisualStudio.TestTools.UnitTesting; + +/// +/// Tests that frontmatter delimiters are recognised only as whole lines at the top of a document. +/// +[TestClass] +public class DelimiterLineTests +{ + private static readonly string Nl = Environment.NewLine; + + [TestMethod] + public void ExtractFrontmatter_BodyHasHorizontalRulesAroundKeyValueLine_LeavesItInTheBody() + { + string input = $"---{Nl}title: A{Nl}---{Nl}Intro{Nl}{Nl}---{Nl}{Nl}note: hi{Nl}{Nl}---{Nl}{Nl}More{Nl}"; + + Dictionary? frontmatter = Frontmatter.ExtractFrontmatter(input); + string body = Frontmatter.ExtractBody(input); + + Assert.IsNotNull(frontmatter); + Assert.HasCount(1, frontmatter); + Assert.AreEqual("A", frontmatter["title"]); + Assert.AreEqual($"Intro{Nl}{Nl}---{Nl}{Nl}note: hi{Nl}{Nl}---{Nl}{Nl}More", body); + } + + [TestMethod] + public void CombineFrontmatter_BodyHasHorizontalRulesAroundKeyValueLine_DoesNotMergeItIntoTheHeader() + { + string input = $"---{Nl}title: A{Nl}---{Nl}Intro{Nl}{Nl}---{Nl}{Nl}note: hi{Nl}{Nl}---{Nl}{Nl}More{Nl}"; + + string result = Frontmatter.CombineFrontmatter(input, FrontmatterNaming.AsIs, FrontmatterOrder.AsIs, FrontmatterMergeStrategy.None); + + Assert.AreEqual($"---{Nl}title: A{Nl}---{Nl}Intro{Nl}{Nl}---{Nl}{Nl}note: hi{Nl}{Nl}---{Nl}{Nl}More{Nl}", result); + } + + [TestMethod] + public void ExtractFrontmatter_ValueContainsDelimiterMidLine_KeepsTheWholeHeader() + { + string input = $"---{Nl}text: foo---{Nl}title: A{Nl}---{Nl}Body{Nl}"; + + Dictionary? frontmatter = Frontmatter.ExtractFrontmatter(input); + string body = Frontmatter.ExtractBody(input); + + Assert.IsNotNull(frontmatter); + Assert.HasCount(2, frontmatter); + Assert.AreEqual("foo---", frontmatter["text"]); + Assert.AreEqual("A", frontmatter["title"]); + Assert.AreEqual("Body", body); + } + + [TestMethod] + public void CombineFrontmatter_TwoConsecutiveBlocks_CombinesThemWithoutLeavingTheSecondInTheBody() + { + string input = $"---{Nl}title: A{Nl}---{Nl}---{Nl}author: B{Nl}---{Nl}Body{Nl}"; + + string result = Frontmatter.CombineFrontmatter(input, FrontmatterNaming.AsIs, FrontmatterOrder.AsIs, FrontmatterMergeStrategy.None); + + Assert.AreEqual($"---{Nl}title: A{Nl}author: B{Nl}---{Nl}Body{Nl}", result); + } + + [TestMethod] + public void ExtractFrontmatter_ClosingDelimiterEndsTheDocument_ReadsTheFrontmatter() + { + string input = $"---{Nl}title: A{Nl}---"; + + Dictionary? frontmatter = Frontmatter.ExtractFrontmatter(input); + + Assert.IsNotNull(frontmatter); + Assert.HasCount(1, frontmatter); + Assert.AreEqual("A", frontmatter["title"]); + } + + [TestMethod] + public void AddFrontmatter_ClosingDelimiterEndsTheDocument_KeepsTheExistingProperties() + { + string input = $"---{Nl}title: A{Nl}---"; + + string result = Frontmatter.AddFrontmatter(input, new Dictionary { ["author"] = "B" }); + Dictionary? frontmatter = Frontmatter.ExtractFrontmatter(result); + + Assert.IsNotNull(frontmatter); + Assert.HasCount(2, frontmatter); + Assert.AreEqual("A", frontmatter["title"]); + Assert.AreEqual("B", frontmatter["author"]); + } + + [TestMethod] + public void AddFrontmatter_ExistingFrontmatterIsUnreadable_ReturnsTheDocumentUnchanged() + { + string input = $"---{Nl}title: [unclosed{Nl}---{Nl}Body{Nl}"; + + string result = Frontmatter.AddFrontmatter(input, new Dictionary { ["author"] = "B" }); + + Assert.AreEqual(input, result); + } + + [TestMethod] + public void AddFrontmatter_ExistingFrontmatterIsEmpty_AddsTheProperties() + { + string input = $"---{Nl}---{Nl}Body{Nl}"; + + string result = Frontmatter.AddFrontmatter(input, new Dictionary { ["author"] = "B" }); + + Assert.AreEqual($"---{Nl}author: B{Nl}---{Nl}Body{Nl}", result); + } +} diff --git a/Frontmatter/Frontmatter.cs b/Frontmatter/Frontmatter.cs index c147943..85bc8ed 100644 --- a/Frontmatter/Frontmatter.cs +++ b/Frontmatter/Frontmatter.cs @@ -171,6 +171,15 @@ public static string AddFrontmatter(string input, Dictionary fro { // Document already has frontmatter, use CombineFrontmatter instead Dictionary? existing = ExtractFrontmatter(input); + + // Frontmatter that is present but could not be read is left alone rather than overwritten, so a + // gap in parsing loses nothing. Only a genuinely empty block is treated as having no properties. + if (existing == null + && (!TrySplitFrontmatterBlocks(input, out List blocks, out _) || blocks.Exists(block => !string.IsNullOrWhiteSpace(block)))) + { + return input; + } + Dictionary combined = CombineFrontmatterObjects(existing ?? [], frontmatter); return ReplaceFrontmatter(input, combined); } @@ -309,20 +318,15 @@ private static string DetectNewLine(string input) private static List> ExtractFrontmatterObjects(string input, out string body) { List> frontmatterObjects = []; - body = input; - if (!HasFrontmatter(input)) + if (!TrySplitFrontmatterBlocks(input, out List blocks, out body)) { return frontmatterObjects; } - string documentNewLine = DetectNewLine(input); - string[] sections = input.Split([FrontmatterDelimiter + documentNewLine], StringSplitOptions.None); - body = string.Join(FrontmatterDelimiter + documentNewLine, sections.Skip(2)); - - for (int i = 1; i < sections.Length; i += 2) + foreach (string block in blocks) { - string section = sections[i].Trim(); + string section = block.Trim(); if (string.IsNullOrWhiteSpace(section)) { continue; @@ -337,6 +341,100 @@ private static List> ExtractFrontmatterObjects(string return frontmatterObjects; } + /// + /// Splits the frontmatter blocks at the top of a document from its body. + /// + /// + /// Delimiters are recognised only as whole lines, so a --- inside a value or a markdown + /// horizontal rule in the body is never mistaken for one. The first line opens a block, which closes + /// at the next delimiter line, including one at the very end of the document. Further blocks are + /// consumed only while each opens on the line straight after the previous one closed; the body is + /// everything after the last block consumed. + /// + /// The markdown document content as a string. + /// The raw text of each frontmatter block, in document order. + /// The document after the last frontmatter block, or the whole input when there is none. + /// True if at least one closed frontmatter block was found, false otherwise. + private static bool TrySplitFrontmatterBlocks(string input, out List blocks, out string body) + { + blocks = []; + body = input; + + if (!HasFrontmatter(input)) + { + return false; + } + + List<(int Start, int End)> lines = SplitLines(input); + bool IsDelimiterAt(int index) => IsDelimiterLine(input[lines[index].Start..lines[index].End]); + + int next = 0; + while (next < lines.Count && IsDelimiterAt(next)) + { + int close = next + 1; + while (close < lines.Count && !IsDelimiterAt(close)) + { + close++; + } + + if (close == lines.Count) + { + break; + } + + blocks.Add(close == next + 1 + ? string.Empty + : input[lines[next + 1].Start..lines[close - 1].End]); + next = close + 1; + } + + if (blocks.Count == 0) + { + return false; + } + + body = next < lines.Count ? input[lines[next].Start..] : string.Empty; + return true; + } + + /// + /// Finds the lines of a document, recognising CRLF, LF and CR endings wherever they occur. + /// + /// The document to split. + /// The start and end offset of each line, excluding its line ending. + private static List<(int Start, int End)> SplitLines(string input) + { + List<(int Start, int End)> lines = []; + int start = 0; + + for (int i = 0; i < input.Length; i++) + { + char c = input[i]; + if (c is not '\r' and not '\n') + { + continue; + } + + lines.Add((start, i)); + if (c == '\r' && i + 1 < input.Length && input[i + 1] == '\n') + { + i++; + } + + start = i + 1; + } + + lines.Add((start, input.Length)); + return lines; + } + + /// + /// Checks whether a line is a frontmatter delimiter, allowing trailing whitespace. + /// + /// The line to check, without its line ending. + /// True if the line is a frontmatter delimiter, false otherwise. + private static bool IsDelimiterLine(string line) => line.TrimEnd() == FrontmatterDelimiter; + /// /// Combines two frontmatter dictionaries into a single dictionary. ///