Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
108 changes: 108 additions & 0 deletions Frontmatter.Test/DecorationOnlyKeyTests.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,108 @@
// Copyright (c) 2023-2026 ktsu-dev contributors

namespace ktsu.Frontmatter.Test;

using System.Collections.Concurrent;
using System.Reflection;

using Microsoft.VisualStudio.TestTools.UnitTesting;

/// <summary>
/// Regression tests for keys that normalize away to nothing — names made only of a decorative
/// prefix, suffix or separator, such as <c>page_</c> or <c>_value</c>.
/// </summary>
/// <remarks>
/// Every non-empty string contains the empty string, so a key whose normalized form is empty is
/// admitted by the containment tests in both the standardizer and the merger against the entire
/// candidate set, and is then renamed to, or merged into, whichever candidate ranks first. The
/// value is silently attributed to an unrelated property. These tests pin the guards that stop it.
/// </remarks>
[TestClass]
public class DecorationOnlyKeyTests
{
[TestInitialize]
public void ClearCaches()
{
ClearCache(typeof(NameStandardizer), "PropertyNameCache");
ClearCache(typeof(PropertyMerger), "PropertyMergeCache");
}

private static void ClearCache(Type type, string fieldName)
{
FieldInfo? field = type.GetField(fieldName, BindingFlags.NonPublic | BindingFlags.Static);
if (field?.GetValue(null) is ConcurrentDictionary<string, string> cache)
{
cache.Clear();
}
}

[TestMethod]
public void StandardizePropertyNames_PreservesDecorationOnlyKeys()
{
foreach (string key in new[] { "page_", "_value", "_", "page__value", "meta_", "custom_" })
{
Dictionary<string, object> result =
NameStandardizer.StandardizePropertyNames(new Dictionary<string, object> { [key] = "V" });

Assert.IsTrue(result.ContainsKey(key), $"[{key}] should be preserved, got [{string.Join(",", result.Keys)}]");
Assert.AreEqual("V", result[key]);
}
}

[TestMethod]
public void MergeSimilarProperties_DoesNotMergeDecorationOnlyKeyIntoAnother()
{
Dictionary<string, object> frontmatter = new()
{
["page_"] = "Decoration",
["description"] = "Real",
};

Dictionary<string, object> result =
PropertyMerger.MergeSimilarProperties(frontmatter, FrontmatterMergeStrategy.Maximum);

Assert.IsTrue(result.ContainsKey("page_"), $"[page_] should survive, got [{string.Join(",", result.Keys)}]");
Assert.AreEqual("Decoration", result["page_"]);
Assert.AreEqual("Real", result["description"]);
}

/// <summary>
/// A padded key and its unpadded form must reach the same decision: either both match the same
/// standard property, or neither matches and each is preserved as written. They did not before —
/// the standardizer lower-cased without trimming and replaced only the space character, so
/// padding blocked the prefix/suffix strip and sent the two forms down different paths.
/// </summary>
/// <remarks>
/// An unmatched key is preserved verbatim, padding included, which is the correct behaviour for
/// a round-tripping library — so the invariant is over the decision, not the literal spelling.
/// </remarks>
[TestMethod]
public void StandardizePropertyNames_TreatsPaddedAndUnpaddedKeysAlike()
{
foreach ((string padded, string bare) in new[]
{
("\tmeta_x_field", "meta_x_field"),
(" page_title ", "page_title"),
("\tuser_name_value\t", "user_name_value"),
(" post_headline ", "post_headline"),
})
{
ClearCaches();
string paddedResult = Single(NameStandardizer.StandardizePropertyNames(new Dictionary<string, object> { [padded] = "V" }));
ClearCaches();
string bareResult = Single(NameStandardizer.StandardizePropertyNames(new Dictionary<string, object> { [bare] = "V" }));

// When the bare key went unmatched it comes back as itself; the padded key should then
// come back as itself too. Otherwise both should land on the same standard property.
string expected = string.Equals(bareResult, bare, StringComparison.Ordinal) ? padded : bareResult;

Assert.AreEqual(expected, paddedResult, $"padded [{padded}] and bare [{bare}] disagree");
}
}

private static string Single(Dictionary<string, object> result)
{
Assert.HasCount(1, result);
return result.Keys.First();
}
}
72 changes: 72 additions & 0 deletions Frontmatter.Test/PropertyNameNormalizerTests.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,72 @@
// Copyright (c) 2023-2026 ktsu-dev contributors

namespace ktsu.Frontmatter.Test;

using Microsoft.VisualStudio.TestTools.UnitTesting;

/// <summary>
/// Tests for <see cref="PropertyNameNormalizer"/>, the single normalization used by both
/// <see cref="NameStandardizer"/> and <see cref="PropertyMerger"/>.
/// </summary>
[TestClass]
public class PropertyNameNormalizerTests
{
private static readonly string[] reviewStatusWords = ["review", "status"];
private static readonly string[] noWords = [];

[TestMethod]
public void Normalize_LowercasesAndCollapsesSeparators()
{
Assert.AreEqual("my_key", PropertyNameNormalizer.Normalize("My-Key"));
Assert.AreEqual("my_key", PropertyNameNormalizer.Normalize("my key"));
Assert.AreEqual("my_key", PropertyNameNormalizer.Normalize("my__key"));
Assert.AreEqual("my_key", PropertyNameNormalizer.Normalize("_my_key_"));
}

[TestMethod]
public void Normalize_StripsOneDecorativePrefixAndSuffix()
{
Assert.AreEqual("title", PropertyNameNormalizer.Normalize("page_title"));
Assert.AreEqual("note", PropertyNameNormalizer.Normalize("note_value"));
Assert.AreEqual("note", PropertyNameNormalizer.Normalize("custom_note_field"));
}

/// <summary>
/// Whitespace is removed before the prefix and suffix strip, so a padded key normalizes to the
/// same thing as its unpadded form. The standardizer's former copy of this logic lower-cased
/// without trimming and replaced only the space character, so a padded — and especially a
/// tab-padded — key kept its padding and missed the strip entirely.
/// </summary>
[TestMethod]
public void Normalize_TrimsAllWhitespaceBeforeStrippingAffixes()
{
foreach (string padded in new[] { " page_title ", "\tpage_title\t", "\npage_title\n", " page_title " })
{
Assert.AreEqual("title", PropertyNameNormalizer.Normalize(padded), $"for [{padded}]");
}

Assert.AreEqual("name", PropertyNameNormalizer.Normalize("\tuser_name_value\t"));
Assert.AreEqual("headline", PropertyNameNormalizer.Normalize(" post_headline "));
}

/// <summary>
/// A key made only of decoration normalizes to the empty string. Both callers must treat that as
/// "nothing to match on" rather than as a value to compare, because every string contains the
/// empty string.
/// </summary>
[TestMethod]
public void Normalize_DecorationOnlyKeysNormalizeToEmpty()
{
foreach (string decoration in new[] { "page_", "_value", "_", "-", " ", "\t", "", "page__value" })
{
Assert.AreEqual(string.Empty, PropertyNameNormalizer.Normalize(decoration), $"for [{decoration}]");
}
}

[TestMethod]
public void NormalizeToWords_SplitsNormalizedForm()
{
Assert.AreSequenceEqual(reviewStatusWords, PropertyNameNormalizer.NormalizeToWords(" page_Review-Status "));
Assert.AreSequenceEqual(noWords, PropertyNameNormalizer.NormalizeToWords("page_"));
}
}
50 changes: 12 additions & 38 deletions Frontmatter/NameStandardizer.cs
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,17 @@ internal static Dictionary<string, object> StandardizePropertyNames(Dictionary<s

private static string? FindStandardPropertyMatch(string normalizedKey, string[] standardProperties)
{
// A key whose normalized form is empty carries nothing to match on: every non-empty standard
// property trivially contains the empty string, so without this guard such a key is admitted
// by the containment test below against the whole standard set and silently renamed to
// whichever one ranks first. Keys that normalize away entirely are decoration-only ("page_",
// "_value", "_") or whitespace, and preserving them is the only safe answer for a library
// whose job is round-tripping frontmatter.
if (normalizedKey.Length == 0)
{
return null;
}

// Try exact match first
string? exactMatch = standardProperties.FirstOrDefault(p =>
string.Equals(NormalizePropertyName(p), normalizedKey, StringComparison.OrdinalIgnoreCase));
Expand Down Expand Up @@ -124,42 +135,5 @@ internal static Dictionary<string, object> StandardizePropertyNames(Dictionary<s
return bestProperty;
}

private static string NormalizePropertyName(string key)
{
// Convert to lowercase
key = key.ToLowerInvariant();

// Remove common prefixes
string[] prefixes = ["page_", "post_", "meta_", "custom_", "user_", "site_"];
foreach (string prefix in prefixes)
{
if (key.StartsWith(prefix))
{
key = key[prefix.Length..];
break;
}
}

// Remove common suffixes
string[] suffixes = ["_value", "_text", "_data", "_info", "_meta", "_field"];
foreach (string suffix in suffixes)
{
if (key.EndsWith(suffix))
{
key = key[..^suffix.Length];
break;
}
}

// Replace special characters with underscores
key = key.Replace('-', '_').Replace(' ', '_');

// Remove duplicate underscores
while (key.Contains("__"))
{
key = key.Replace("__", "_");
}

return key.Trim('_');
}
private static string NormalizePropertyName(string key) => PropertyNameNormalizer.Normalize(key);
}
47 changes: 13 additions & 34 deletions Frontmatter/PropertyMerger.cs
Original file line number Diff line number Diff line change
Expand Up @@ -103,7 +103,7 @@

return !isKnownMapping && !hasExactMatch && !isInSameCategory
? key
: PropertyMappings.All.TryGetValue(canonicalName, out string? knownName) ? knownName : canonicalName;

Check warning on line 106 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Extract this nested ternary operation into an independent statement.

Check warning on line 106 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Extract this nested ternary operation into an independent statement.

Check warning on line 106 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Extract this nested ternary operation into an independent statement.

Check warning on line 106 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Extract this nested ternary operation into an independent statement.
}

private static bool IsInSameCategory(string key, string canonicalName)
Expand Down Expand Up @@ -195,7 +195,7 @@
/// <param name="key">The key to analyze.</param>
/// <param name="existingKeys">All existing keys in the frontmatter.</param>
/// <returns>The canonical name for the key.</returns>
private static string FindBasicCanonicalName(string key, string[] existingKeys)

Check warning on line 198 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Refactor this method to reduce its Cognitive Complexity from 19 to the 15 allowed.

Check warning on line 198 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Refactor this method to reduce its Cognitive Complexity from 19 to the 15 allowed.

Check warning on line 198 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Refactor this method to reduce its Cognitive Complexity from 19 to the 15 allowed.

Check warning on line 198 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Refactor this method to reduce its Cognitive Complexity from 19 to the 15 allowed.

Check warning on line 198 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Refactor this method to reduce its Cognitive Complexity from 19 to the 15 allowed.

Check warning on line 198 in Frontmatter/PropertyMerger.cs

View workflow job for this annotation

GitHub Actions / Analyze & Release

Refactor this method to reduce its Cognitive Complexity from 19 to the 15 allowed.
{
// First check if it's a known property
if (PropertyMappings.All.TryGetValue(key, out string? canonicalName))
Expand All @@ -206,6 +206,16 @@
// Remove common prefixes/suffixes and special characters
string normalizedKey = NormalizePropertyName(key);

// A key that normalizes away to nothing has no content to match on. Both loops below would
// otherwise treat it as equal to, or contained by, every other key — the exact-match loop
// pairs it with any other decoration-only key, and the partial-match loop admits it against
// all of them, since every string contains the empty string. Either way the key is merged
// into an unrelated one and its value is lost, so leave it alone.
if (normalizedKey.Length == 0)
{
return key;
}

// Look for exact matches after normalization
foreach (string existingKey in existingKeys)
{
Expand Down Expand Up @@ -258,15 +268,15 @@
}

// Then try more aggressive matching using word similarity
string[] keyWords = NormalizePropertyName(key).Split(['-', ' ', '_'], StringSplitOptions.RemoveEmptyEntries);
string[] keyWords = PropertyNameNormalizer.NormalizeToWords(key);

// Find best match among existing keys
(string Key, int Score)? bestMatch = existingKeys
.Select(existingKey => (
Key: existingKey,
Score: CalculateWordMatchScore(
keyWords,
NormalizePropertyName(existingKey).Split(['-', ' ', '_'], StringSplitOptions.RemoveEmptyEntries)
PropertyNameNormalizer.NormalizeToWords(existingKey)
)
))
.Where(match => match.Score > 0)
Expand All @@ -277,38 +287,7 @@
return bestMatch?.Key ?? key;
}

private static string NormalizePropertyName(string key)
{
// Convert to lowercase and trim
key = key.Trim().ToLowerInvariant();

// Common prefixes and suffixes to remove
string[] prefixes = ["page_", "post_", "meta_", "custom_", "user_", "site_"];
string[] suffixes = ["_value", "_text", "_data", "_info", "_meta", "_field"];

// Remove prefixes
foreach (string prefix in prefixes)
{
if (key.StartsWith(prefix, StringComparison.Ordinal))
{
key = key[prefix.Length..];
break;
}
}

// Remove suffixes
foreach (string suffix in suffixes)
{
if (key.EndsWith(suffix, StringComparison.Ordinal))
{
key = key[..^suffix.Length];
break;
}
}

// Replace special characters with underscores and remove duplicates
return string.Join("_", key.Split(['-', ' ', '_'], StringSplitOptions.RemoveEmptyEntries));
}
private static string NormalizePropertyName(string key) => PropertyNameNormalizer.Normalize(key);

private static int CalculateWordMatchScore(string[] words1, string[] words2)
{
Expand Down
79 changes: 79 additions & 0 deletions Frontmatter/PropertyNameNormalizer.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
// Copyright (c) 2023-2026 ktsu-dev contributors

namespace ktsu.Frontmatter;

/// <summary>
/// Normalizes frontmatter property names into a common form so that keys which differ only by
/// casing, separators, surrounding whitespace or a decorative prefix/suffix compare equal.
/// </summary>
/// <remarks>
/// <para>
/// <see cref="NameStandardizer"/> and <see cref="PropertyMerger"/> each carried their own copy of
/// this logic. The two had drifted apart: the standardizer lower-cased without trimming, stripped
/// its prefix and suffix with culture-sensitive comparisons, and replaced only the space character,
/// so a key carrying leading or trailing whitespace — a tab especially — kept it and missed the
/// prefix/suffix strip entirely. This single implementation takes the merger's behaviour, which is
/// the more robust of the two on every point of difference.
/// </para>
/// <para>
/// Trimming happens before the prefix and suffix strip so that <c>" page_title "</c> normalizes the
/// same as <c>"page_title"</c>, and <see cref="string.Trim()"/> covers every whitespace character
/// rather than the space alone. The prefix and suffix comparisons are ordinal because these are
/// fixed ASCII tokens, matched against an already invariantly-lowercased key.
/// </para>
/// </remarks>
internal static class PropertyNameNormalizer
{
/// <summary>
/// Decorative leading tokens that carry no meaning for matching purposes.
/// </summary>
private static readonly string[] Prefixes = ["page_", "post_", "meta_", "custom_", "user_", "site_"];

/// <summary>
/// Decorative trailing tokens that carry no meaning for matching purposes.
/// </summary>
private static readonly string[] Suffixes = ["_value", "_text", "_data", "_info", "_meta", "_field"];

/// <summary>
/// Characters treated as word separators within a property name.
/// </summary>
private static readonly char[] Separators = ['-', ' ', '_'];

/// <summary>
/// Normalizes a property name for comparison.
/// </summary>
/// <param name="key">The property name to normalize.</param>
/// <returns>
/// The name lower-cased and trimmed, with at most one decorative prefix and one decorative
/// suffix removed, and all separator runs collapsed to a single underscore. Leading and
/// trailing separators are dropped, so the result never begins or ends with an underscore.
/// </returns>
internal static string Normalize(string key)
{
key = key.Trim().ToLowerInvariant();

// At most one prefix and one suffix are removed, so this is find-first-then-act rather than
// a filter: the captured key is reassigned between the two steps.
string? matchedPrefix = Array.Find(Prefixes, prefix => key.StartsWith(prefix, StringComparison.Ordinal));
if (matchedPrefix is not null)
{
key = key[matchedPrefix.Length..];
}

string? matchedSuffix = Array.Find(Suffixes, suffix => key.EndsWith(suffix, StringComparison.Ordinal));
if (matchedSuffix is not null)
{
key = key[..^matchedSuffix.Length];
}

return string.Join("_", key.Split(Separators, StringSplitOptions.RemoveEmptyEntries));
}

/// <summary>
/// Normalizes a property name and splits it into its constituent words.
/// </summary>
/// <param name="key">The property name to normalize and split.</param>
/// <returns>The normalized name's words, with empty entries removed.</returns>
internal static string[] NormalizeToWords(string key) =>
Normalize(key).Split(Separators, StringSplitOptions.RemoveEmptyEntries);
}
Loading