diff --git a/Frontmatter.Test/StackedBlockPreservationTests.cs b/Frontmatter.Test/StackedBlockPreservationTests.cs new file mode 100644 index 0000000..480af95 --- /dev/null +++ b/Frontmatter.Test/StackedBlockPreservationTests.cs @@ -0,0 +1,72 @@ +// Copyright (c) 2023-2026 ktsu-dev contributors + +namespace ktsu.Frontmatter.Test; + +using System.Collections.Generic; + +using Microsoft.VisualStudio.TestTools.UnitTesting; + +/// +/// Regression tests for documents with more than one stacked frontmatter block. The rebuilt header used +/// to be built from a subset of the blocks while the body still started after all of them, so the +/// blocks left out were deleted from the document. +/// +[TestClass] +public class StackedBlockPreservationTests +{ + [TestMethod] + public void AddFrontmatter_TwoStackedBlocks_KeepsPropertiesFromEveryBlock() + { + const string input = "---\ntitle: T\n---\n---\ntags: [x]\n---\nbody\n"; + + string result = Frontmatter.AddFrontmatter(input, new() { ["author"] = "Me" }); + + Dictionary? frontmatter = Frontmatter.ExtractFrontmatter(result); + Assert.IsNotNull(frontmatter); + Assert.AreEqual("T", frontmatter["title"]); + Assert.AreEqual("Me", frontmatter["author"]); + Assert.IsTrue(frontmatter.ContainsKey("tags"), "The second block's tags should survive adding a property"); + Assert.AreEqual("body", Frontmatter.ExtractBody(result)); + } + + [TestMethod] + public void AddFrontmatter_StackedBlocksRepeatAKey_FirstBlockWins() + { + const string input = "---\ntitle: First\n---\n---\ntitle: Second\n---\nbody\n"; + + string result = Frontmatter.AddFrontmatter(input, new() { ["author"] = "Me" }); + + Dictionary? frontmatter = Frontmatter.ExtractFrontmatter(result); + Assert.IsNotNull(frontmatter); + Assert.AreEqual("First", frontmatter["title"]); + } + + // The first block is always read as frontmatter, so an unparseable first block followed by one that + // parses is the case where the unreadable text sits in the header rather than the body. + private const string UnreadableFirstBlock = "---\nkey: [unclosed\n---\n---\ntitle: T\n---\nbody\n"; + + [TestMethod] + public void AddFrontmatter_UnreadableFirstBlock_ReturnsInputUnchanged() => + Assert.AreEqual(UnreadableFirstBlock, Frontmatter.AddFrontmatter(UnreadableFirstBlock, new() { ["author"] = "Me" })); + + [TestMethod] + [DataRow(FrontmatterMergeStrategy.None)] + [DataRow(FrontmatterMergeStrategy.Conservative)] + [DataRow(FrontmatterMergeStrategy.Maximum)] + public void CombineFrontmatter_UnreadableFirstBlock_ReturnsInputUnchanged(FrontmatterMergeStrategy strategy) + { + string result = Frontmatter.CombineFrontmatter(UnreadableFirstBlock, FrontmatterNaming.Standard, FrontmatterOrder.Sorted, strategy); + + Assert.AreEqual(UnreadableFirstBlock, result); + } + + [TestMethod] + public void CombineFrontmatter_UnreadableSecondBlock_StaysVerbatimInTheBody() + { + const string input = "---\ntitle: T\n---\n---\nkey: [unclosed\n---\nbody\n"; + + string result = Frontmatter.CombineFrontmatter(input); + + Assert.Contains("key: [unclosed", result); + } +} diff --git a/Frontmatter/Frontmatter.cs b/Frontmatter/Frontmatter.cs index fcdb2cd..a9e5575 100644 --- a/Frontmatter/Frontmatter.cs +++ b/Frontmatter/Frontmatter.cs @@ -78,20 +78,18 @@ public static string CombineFrontmatter(string input, FrontmatterNaming property return cachedResult; } - List> frontmatterObjects = ExtractFrontmatterObjects(input, out string body); + List> frontmatterObjects = ExtractFrontmatterObjects(input, out string body, out bool hasUnreadableBlock); - if (frontmatterObjects.Count == 0) + // A block that could not be parsed would be dropped from the rebuilt header while the body still + // starts after it, so the document is left alone rather than losing that block's text. + if (frontmatterObjects.Count == 0 || hasUnreadableBlock) { // Cache the original content since no processing was needed ProcessedFrontmatterCache.TryAdd(cacheKey, input); return input; } - Dictionary combinedFrontmatterObject = frontmatterObjects.First(); - foreach (Dictionary frontmatterObject in frontmatterObjects.Skip(1)) - { - combinedFrontmatterObject = CombineFrontmatterObjects(combinedFrontmatterObject, frontmatterObject); - } + Dictionary combinedFrontmatterObject = CombineAllFrontmatterObjects(frontmatterObjects); // Apply property merging if enabled if (mergeStrategy != FrontmatterMergeStrategy.None) @@ -184,18 +182,17 @@ public static string AddFrontmatter(string input, Dictionary fro if (HasFrontmatter(input)) { - // Document already has frontmatter, use CombineFrontmatter instead - Dictionary? existing = ExtractFrontmatter(input); + // Every existing block is folded in, since ReplaceFrontmatter keeps only the body after all of them. + List> existing = ExtractFrontmatterObjects(input, out _, out bool hasUnreadableBlock); // Frontmatter that is present but could not be read is left alone rather than overwritten, so a // gap in parsing loses nothing. Only a genuinely empty block is treated as having no properties. - if (existing == null - && (!TrySplitFrontmatterBlocks(input, out List blocks, out _) || blocks.Exists(block => !string.IsNullOrWhiteSpace(block)))) + if (hasUnreadableBlock) { return input; } - Dictionary combined = CombineFrontmatterObjects(existing ?? [], frontmatter); + Dictionary combined = CombineFrontmatterObjects(CombineAllFrontmatterObjects(existing), frontmatter); return ReplaceFrontmatter(input, combined); } @@ -304,12 +301,27 @@ private static Dictionary SortFrontmatterProperties(DictionaryOutput parameter that will contain the markdown body without frontmatter. /// A collection of dictionaries representing each frontmatter section. /// Thrown when there are too many frontmatter sections in the document. - private static List> ExtractFrontmatterObjects(string input, out string body) + private static List> ExtractFrontmatterObjects(string input, out string body) => + ExtractFrontmatterObjects(input, out body, out _); + + /// + /// Extracts all frontmatter objects from a markdown document, reporting whether any of it could not be read. + /// + /// The markdown document content as a string. + /// Output parameter that will contain the markdown body without frontmatter. + /// + /// True when the document opens with a frontmatter delimiter but a non-blank block failed to parse, or no + /// closed block could be found. Such text is missing from the returned objects but not from the document. + /// + /// A collection of dictionaries representing each frontmatter section that parsed. + private static List> ExtractFrontmatterObjects(string input, out string body, out bool hasUnreadableBlock) { List> frontmatterObjects = []; + hasUnreadableBlock = false; if (!TrySplitFrontmatterBlocks(input, out List blocks, out body)) { + hasUnreadableBlock = HasFrontmatter(input); return frontmatterObjects; } @@ -325,11 +337,31 @@ private static List> ExtractFrontmatterObjects(string { frontmatterObjects.Add(frontmatterObject); } + else + { + hasUnreadableBlock = true; + } } return frontmatterObjects; } + /// + /// Folds frontmatter objects into one, in document order, so an earlier block wins a repeated key. + /// + /// The frontmatter objects to fold. + /// The combined frontmatter, empty when there are no objects. + private static Dictionary CombineAllFrontmatterObjects(List> frontmatterObjects) + { + Dictionary combined = []; + foreach (Dictionary frontmatterObject in frontmatterObjects) + { + combined = CombineFrontmatterObjects(combined, frontmatterObject); + } + + return combined; + } + /// /// Splits the frontmatter blocks at the top of a document from its body. ///