From a067aac62c5e3393a8680ed8d8024ca3c9bc6ad8 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 25 Sep 2026 08:33:33 +0000 Subject: [PATCH] [patch] Stop a single-character key being containment-matched onto unrelated properties A key like meta_x_field normalizes to the single character x, and the containment tests in NameStandardizer and PropertyMerger are bidirectional, so x is admitted against any candidate that happens to use that letter: "next" contains "x". The key is then renamed, or merged, onto an unrelated property. Both classes already guarded the empty normalized form, with comments describing exactly this failure. A length of 1 fell through that guard. Measured on main: StandardizePropertyNames turned meta_x_field into next, page_a_value into area, meta_e_field into editor, custom_s_data into slug and page_t_data into toc. MergeSimilarProperties was worse than reattribution -- {meta_x_field, next, text} came back as {meta_x_field, next}, dropping text entirely, because containment matched mutually in both directions. The rule now lives once, as PropertyNameNormalizer.MayMatchByContainment, and applies to the three containment sites: the standardizer's partial-match loop, the merger's partial-match loop, and CalculateWordMatchScore's containment branch, which could otherwise re-admit the same key through the word-score path. Containment only. Exact matching on a one-character normalized form is untouched, and the floor is two characters rather than more, so the shortest real names anywhere in StandardOrder.PropertyNames or PropertyMappings -- by, tag, url -- still match as before. Fixes ktsu-dev/Frontmatter#128 Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_013qMsnuVCPrfVynzgpbKQTD --- Frontmatter.Test/SingleCharacterKeyTests.cs | 123 ++++++++++++++++++++ Frontmatter/NameStandardizer.cs | 5 + Frontmatter/PropertyMerger.cs | 8 +- Frontmatter/PropertyNameNormalizer.cs | 35 ++++++ 4 files changed, 170 insertions(+), 1 deletion(-) create mode 100644 Frontmatter.Test/SingleCharacterKeyTests.cs diff --git a/Frontmatter.Test/SingleCharacterKeyTests.cs b/Frontmatter.Test/SingleCharacterKeyTests.cs new file mode 100644 index 0000000..4ca9b14 --- /dev/null +++ b/Frontmatter.Test/SingleCharacterKeyTests.cs @@ -0,0 +1,123 @@ +// Copyright (c) 2023-2026 ktsu-dev contributors + +namespace ktsu.Frontmatter.Test; + +using System.Collections.Concurrent; +using System.Reflection; + +using Microsoft.VisualStudio.TestTools.UnitTesting; + +/// +/// Regression tests for keys whose normalized form is a single character, such as +/// meta_x_field — a decorative prefix and suffix around one letter. +/// +/// +/// +/// A one-character fragment is contained in any candidate that happens to use that letter, so the +/// bidirectional containment tests in the standardizer and the merger admit it on an incidental +/// letter rather than a shared word. meta_x_field normalizes to x, and next +/// contains x. +/// +/// +/// This is the same failure pins for the empty normalized +/// form, one character further along: the value is silently attributed to an unrelated property, +/// and in the merger's case a third property is dropped outright. These tests pin the guards. +/// +/// +[TestClass] +public class SingleCharacterKeyTests +{ + [TestInitialize] + public void ClearCaches() + { + ClearCache(typeof(NameStandardizer), "PropertyNameCache"); + ClearCache(typeof(PropertyMerger), "PropertyMergeCache"); + } + + private static void ClearCache(Type type, string fieldName) + { + FieldInfo? field = type.GetField(fieldName, BindingFlags.NonPublic | BindingFlags.Static); + if (field?.GetValue(null) is ConcurrentDictionary cache) + { + cache.Clear(); + } + } + + /// + /// Each of these normalizes to one character and was renamed to an unrelated standard property: + /// x to next, a to area, e to editor, s to + /// slug and t to toc. + /// + [TestMethod] + public void StandardizePropertyNames_PreservesKeysNormalizingToOneCharacter() + { + foreach (string key in new[] { "meta_x_field", "page_a_value", "meta_e_field", "custom_s_data", "page_x_text", "page_t_data" }) + { + ClearCaches(); + + Dictionary result = + NameStandardizer.StandardizePropertyNames(new Dictionary { [key] = "V" }); + + Assert.IsTrue(result.ContainsKey(key), $"[{key}] should be preserved, got [{string.Join(",", result.Keys)}]"); + Assert.AreEqual("V", result[key]); + } + } + + /// + /// The merger cross-referenced mutually — meta_x_field to next and next back + /// to meta_x_field, since containment is tested in both directions — and a third key that + /// also contained the letter was dropped entirely rather than merged. + /// + [TestMethod] + public void MergeSimilarProperties_DoesNotMergeOrDropOnASingleCharacterFragment() + { + Dictionary frontmatter = new() + { + ["meta_x_field"] = "A", + ["next"] = "B", + ["text"] = "C", + }; + + Dictionary result = + PropertyMerger.MergeSimilarProperties(frontmatter, FrontmatterMergeStrategy.Maximum); + + Assert.HasCount(3, result, $"nothing should be dropped, got [{string.Join(",", result.Keys)}]"); + Assert.AreEqual("A", result["meta_x_field"]); + Assert.AreEqual("B", result["next"]); + Assert.AreEqual("C", result["text"]); + } + + /// + /// The rule is about containment specifically, and it is symmetric: a one-character name is + /// unmatchable whichever side it appears on, because containment is tested in both directions. + /// + [TestMethod] + public void MayMatchByContainment_RejectsAOneCharacterNameOnEitherSide() + { + Assert.IsFalse(PropertyNameNormalizer.MayMatchByContainment("x", "next"), "short first"); + Assert.IsFalse(PropertyNameNormalizer.MayMatchByContainment("next", "x"), "short second"); + Assert.IsFalse(PropertyNameNormalizer.MayMatchByContainment("x", "y"), "both short"); + Assert.IsFalse(PropertyNameNormalizer.MayMatchByContainment("", "next"), "empty is also below the floor"); + + Assert.IsTrue(PropertyNameNormalizer.MayMatchByContainment("by", "author"), "two characters is the floor"); + Assert.IsTrue(PropertyNameNormalizer.MayMatchByContainment("tag", "tags")); + Assert.IsTrue(PropertyNameNormalizer.MayMatchByContainment("url", "canonical_url")); + } + + /// + /// The floor is two characters rather than something larger, so a key normalizing to a + /// two-character fragment still matches by containment exactly as it did before. ag is + /// contained in tags, stage and image. + /// + [TestMethod] + public void StandardizePropertyNames_StillMatchesATwoCharacterFragment() + { + ClearCaches(); + + Dictionary result = + NameStandardizer.StandardizePropertyNames(new Dictionary { ["meta_ag_field"] = "V" }); + + Assert.IsFalse(result.ContainsKey("meta_ag_field"), + $"[ag] is at the floor and should still be standardized, got [{string.Join(",", result.Keys)}]"); + } +} diff --git a/Frontmatter/NameStandardizer.cs b/Frontmatter/NameStandardizer.cs index 3686b1a..b9c69bc 100644 --- a/Frontmatter/NameStandardizer.cs +++ b/Frontmatter/NameStandardizer.cs @@ -119,6 +119,11 @@ internal static Dictionary StandardizePropertyNames(Dictionary + /// The shortest normalized name a containment test is allowed to match on. + /// + /// + /// Two is the smallest value that changes nothing legitimate: the shortest names anywhere in + /// or are by, + /// tag and url, and no standard property or mapping name normalizes to fewer than + /// two characters, so nothing real is matched by containment on a single character. + /// + private const int MinimumMatchableLength = 2; + + /// + /// Whether two normalized names carry enough content to be compared by containment. + /// + /// One normalized name. + /// The other normalized name. + /// when a containment match between the two would be meaningful. + /// + /// + /// Containment is tested in both directions, so the short side is what makes a match meaningless + /// regardless of which argument it is. A one-character fragment is contained in every candidate + /// that happens to use that letter, so it is admitted on an incidental letter rather than on a + /// shared word: meta_x_field normalizes to x, and next contains x. + /// + /// + /// This is the same failure the empty-string guards in and + /// already document, one character further along — a value silently + /// attributed to an unrelated property, and in the merger's case dropped outright. It applies to + /// containment only: exact matching on a one-character normalized form stays available, so a + /// genuine single-character key still pairs with another key that normalizes to the same thing. + /// + /// + internal static bool MayMatchByContainment(string first, string second) => + first.Length >= MinimumMatchableLength && second.Length >= MinimumMatchableLength; + /// /// Normalizes a property name and splits it into its constituent words. ///