From 9292d3c267375dd464ae981bdeb2227b07134ae6 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Wed, 30 Apr 2025 14:56:25 -0700 Subject: [PATCH 01/13] Improve the emphasis regex. I was getting failures where URLs with underscores were getting blocks thrown in them, creating invalid HTML. --- include/maddy/emphasizedparser.h | 4 +- tests/maddy/test_maddy_emphasizedparser.cpp | 109 ++++++++++++++++++++ 2 files changed, 112 insertions(+), 1 deletion(-) diff --git a/include/maddy/emphasizedparser.h b/include/maddy/emphasizedparser.h index 1705e6f..af30928 100644 --- a/include/maddy/emphasizedparser.h +++ b/include/maddy/emphasizedparser.h @@ -40,8 +40,10 @@ class EmphasizedParser : public LineParser */ void Parse(std::string& line) override { + // Modifed from previous version, with help from + // https://stackoverflow.com/questions/61346949/regex-for-markdown-emphasis static std::regex re( - R"((?!.*`.*|.*.*)_(?!.*`.*|.*<\/code>.*)([^_]*)_(?!.*`.*|.*<\/code>.*))" + R"((?!.*`.*|.*.*)\b_(?![\s])(?!.*`.*|.*<\/code>.*)(.*?[^\s])_\b(?!.*`.*|.*<\/code>.*))" ); static std::string replacement = "$1"; diff --git a/tests/maddy/test_maddy_emphasizedparser.cpp b/tests/maddy/test_maddy_emphasizedparser.cpp index 6442779..8da0b4f 100644 --- a/tests/maddy/test_maddy_emphasizedparser.cpp +++ b/tests/maddy/test_maddy_emphasizedparser.cpp @@ -21,6 +21,85 @@ TEST(MADDY_EMPHASIZEDPARSER, ItReplacesMarkdownWithEmphasizedHTML) ASSERT_EQ(expected, text); } +TEST(MADDY_EMPHASIZEDPARSER, ItReplacesUnderscoresAtStringEdges) +{ + std::string text = "_some text_"; + std::string expected = "some text"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItDoesNotReplaceMarkdownWithInlineUnderscores) +{ + std::string text = "some text_bla_text testing _it_ out"; + std::string expected = "some text_bla_text testing it out"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItOnlyReplacesUnderscoresAtWordBreaks) +{ + std::string text = "some _text_bla_ testing _it_ out"; + std::string expected = "some text_bla testing it out"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItReplacesUnderscoresWithMultipleWords) +{ + std::string text = "some _text testing it_ out"; + std::string expected = "some text testing it out"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItAllowsDoubleUnderscores) +{ + // I'm not sure if this is standard or not, but this is how the github markdown + // parser behaves. Other things I've seen want it to *not* match. + std::string text = "some __text testing it_ out"; + std::string expected = "some _text testing it out"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItDoesntReplaceUnderscoresInsideCodeBlocks) +{ + std::string text = "Stuff inside blocks _shouldn't be emphasized_ at all"; + std::string expected = "Stuff inside blocks _shouldn't be emphasized_ at all"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItDoesNotReplaceUnderscoresInURLs) +{ + std::string text = "[Link Title](http://example.com/what_you_didn't_know)"; + std::string expected = "[Link Title](http://example.com/what_you_didn't_know)"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + TEST(MADDY_EMPHASIZEDPARSER, ItDoesNotParseInsideInlineCode) { std::string text = "some text `*bla*` `/**text*/` testing _it_ out"; @@ -32,3 +111,33 @@ TEST(MADDY_EMPHASIZEDPARSER, ItDoesNotParseInsideInlineCode) ASSERT_EQ(expected, text); } + +TEST(MADDY_EMPHASIZEDPARSER, ItParsesOutsideCodeBlocks) +{ + std::string text = + "Stuff inside blocks _shouldn't be emphasized_ " + " but outside _should_."; + std::string expected = + "Stuff inside blocks _shouldn't be emphasized_ " + " but outside should."; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItParsesOutsideTickBlocks) +{ + std::string text = + "Stuff inside `blocks _shouldn't be emphasized_ `" + " but outside _should_."; + std::string expected = + "Stuff inside `blocks _shouldn't be emphasized_ `" + " but outside should."; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} From 2df2fe7fa6e6a68e199d1c0d3c8c993d94b218eb Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Wed, 30 Apr 2025 16:23:19 -0700 Subject: [PATCH 02/13] Add strong parser tests. Most don't pass (and are disabled), but should. Unfortunately, adding a word boundary (\b) to the strong regex works for these tests, but somehow the full parser then breaks. The regex that I believed should work is added as a comment, for anyone wishing to make things work going forward. --- include/maddy/strongparser.h | 23 ++++- tests/maddy/test_maddy_strongparser.cpp | 113 ++++++++++++++++++++++++ 2 files changed, 134 insertions(+), 2 deletions(-) diff --git a/include/maddy/strongparser.h b/include/maddy/strongparser.h index 348f2d4..56df620 100644 --- a/include/maddy/strongparser.h +++ b/include/maddy/strongparser.h @@ -40,12 +40,31 @@ class StrongParser : public LineParser */ void Parse(std::string& line) override { + // This version of the regex is changed exactly the same way + // that the regex for the emphasized parser was changed, and + // it then passes all the 'disabled' tests in the 'strong parser' + // test, but then it fails general parsing. For some reason, + // "__text__" translates "text" even though there + // are no word boundaries at the correct places. It's weird! + // + //static std::vector res{ + // std::regex{ + // R"((?!.*`.*|.*.*)\b\*\*(?![\s])(?!.*`.*|.*<\/code>.*)" + // "(.*?[^\s])\*\*\b(?!.*`.*|.*<\/code>.*))" + // }, + // std::regex{ + // R"((?!.*`.*|.*.*)\b__(?![\s])(?!.*`.*|.*<\/code>.*)" + // "(.*?[^\s])__\b(?!.*`.*|.*<\/code>.*))" + // } + //}; static std::vector res{ std::regex{ - R"((?!.*`.*|.*.*)\*\*(?!.*`.*|.*<\/code>.*)([^\*\*]*)\*\*(?!.*`.*|.*<\/code>.*))" + R"((?!.*`.*|.*.*)\*\*(?!.*`.*|.*<\/code>.*)" + "([^\*\*]*)\*\*(?!.*`.*|.*<\/code>.*))" }, std::regex{ - R"((?!.*`.*|.*.*)__(?!.*`.*|.*<\/code>.*)([^__]*)__(?!.*`.*|.*<\/code>.*))" + R"((?!.*`.*|.*.*)__(?!.*`.*|.*<\/code>.*)" + "([^__]*)__(?!.*`.*|.*<\/code>.*))" } }; static std::string replacement = "$1"; diff --git a/tests/maddy/test_maddy_strongparser.cpp b/tests/maddy/test_maddy_strongparser.cpp index f006e26..5fd962f 100644 --- a/tests/maddy/test_maddy_strongparser.cpp +++ b/tests/maddy/test_maddy_strongparser.cpp @@ -83,3 +83,116 @@ TEST(MADDY_STRONGPARSER, ItDoesNotParseInsideInlineCode) ASSERT_EQ(test.expected, test.text); } } + +TEST(MADDY_STRONGPARSER, ItReplacesUnderscoresAtStringEdges) +{ + std::string text = "__some text__"; + std::string expected = "some text"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(DISABLED_MADDY_STRONGPARSER, ItDoesNotReplaceMarkdownWithInlineUnderscores) +{ + std::string text = "some text__bla__text testing __it__ out"; + std::string expected = "some text__bla__text testing it out"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(DISABLED_MADDY_STRONGPARSER, ItOnlyReplacesUnderscoresAtWordBreaks) +{ + std::string text = "some __text__bla__ testing __it__ out"; + std::string expected = + "some text__bla testing it out"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_STRONGPARSER, ItReplacesUnderscoresWithMultipleWords) +{ + std::string text = "some __text testing it__ out"; + std::string expected = "some text testing it out"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(DISABLED_MADDY_STRONGPARSER, ItAllowsTripleUnderscores) +{ + // I'm not sure if this is standard or not, but this is how the github + // markdown parser behaves. Other things I've seen want it to *not* match. + std::string text = "some ___text testing it__ out"; + std::string expected = "some _text testing it out"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_STRONGPARSER, ItDoesntReplaceUnderscoresInsideCodeBlocks) +{ + std::string text = + "Stuff inside blocks __shouldn't be strong__ at all"; + std::string expected = + "Stuff inside blocks __shouldn't be strong__ at all"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(DISABLED_MADDY_STRONGPARSER, ItDoesNotReplaceUnderscoresInURLs) +{ + std::string text = "[Link Title](http://example.com/what__you__didn't__know)"; + std::string expected = + "[Link Title](http://example.com/what__you__didn't__know)"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_STRONGPARSER, ItParsesOutsideCodeBlocks) +{ + std::string text = + "Stuff inside blocks __shouldn't be strong__ " + " but outside __should__."; + std::string expected = + "Stuff inside blocks __shouldn't be strong__ " + " but outside should."; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_STRONGPARSER, ItParsesOutsideTickBlocks) +{ + std::string text = + "Stuff inside `blocks __shouldn't be strong__ `" + " but outside __should__."; + std::string expected = + "Stuff inside `blocks __shouldn't be strong__ `" + " but outside should."; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} From c89a98386501946ebabbc62e3a2911c4c101950e Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Wed, 30 Apr 2025 16:30:39 -0700 Subject: [PATCH 03/13] Clang fixes; changelog. --- CHANGELOG.md | 1 + include/maddy/strongparser.h | 28 ++++++++++++------------- tests/maddy/test_maddy_strongparser.cpp | 4 +++- 3 files changed, 17 insertions(+), 16 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 63c0fca..25072a3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,7 @@ maddy uses [semver versioning](https://semver.org/). ## Upcoming +* ![**FIXED**](https://img.shields.io/badge/-FIXED-%23090) Do not create emphasis tags not at word boundaries, i.e. `only_internal_underscores`. * ... ## version 1.5.0 2025-04-21 diff --git a/include/maddy/strongparser.h b/include/maddy/strongparser.h index 56df620..5cf9c38 100644 --- a/include/maddy/strongparser.h +++ b/include/maddy/strongparser.h @@ -46,24 +46,22 @@ class StrongParser : public LineParser // test, but then it fails general parsing. For some reason, // "__text__" translates "text" even though there // are no word boundaries at the correct places. It's weird! - // - //static std::vector res{ - // std::regex{ - // R"((?!.*`.*|.*.*)\b\*\*(?![\s])(?!.*`.*|.*<\/code>.*)" - // "(.*?[^\s])\*\*\b(?!.*`.*|.*<\/code>.*))" - // }, - // std::regex{ - // R"((?!.*`.*|.*.*)\b__(?![\s])(?!.*`.*|.*<\/code>.*)" - // "(.*?[^\s])__\b(?!.*`.*|.*<\/code>.*))" - // } - //}; + + // static std::vector res{ + // std::regex{ + // R"((?!.*`.*|.*.*)\b\*\*(?![\s])(?!.*`.*|.*<\/code>.*)" + // "(.*?[^\s])\*\*\b(?!.*`.*|.*<\/code>.*))" + // }, + // std::regex{ + // R"((?!.*`.*|.*.*)\b__(?![\s])(?!.*`.*|.*<\/code>.*)" + // "(.*?[^\s])__\b(?!.*`.*|.*<\/code>.*))" + // } + // }; static std::vector res{ - std::regex{ - R"((?!.*`.*|.*.*)\*\*(?!.*`.*|.*<\/code>.*)" + std::regex{R"((?!.*`.*|.*.*)\*\*(?!.*`.*|.*<\/code>.*)" "([^\*\*]*)\*\*(?!.*`.*|.*<\/code>.*))" }, - std::regex{ - R"((?!.*`.*|.*.*)__(?!.*`.*|.*<\/code>.*)" + std::regex{R"((?!.*`.*|.*.*)__(?!.*`.*|.*<\/code>.*)" "([^__]*)__(?!.*`.*|.*<\/code>.*))" } }; diff --git a/tests/maddy/test_maddy_strongparser.cpp b/tests/maddy/test_maddy_strongparser.cpp index 5fd962f..466068d 100644 --- a/tests/maddy/test_maddy_strongparser.cpp +++ b/tests/maddy/test_maddy_strongparser.cpp @@ -132,7 +132,9 @@ TEST(MADDY_STRONGPARSER, ItReplacesUnderscoresWithMultipleWords) TEST(DISABLED_MADDY_STRONGPARSER, ItAllowsTripleUnderscores) { // I'm not sure if this is standard or not, but this is how the github - // markdown parser behaves. Other things I've seen want it to *not* match. + // markdown parser behaves. Other things I've seen want it to *not* + // match. + std::string text = "some ___text testing it__ out"; std::string expected = "some _text testing it out"; auto strongParser = std::make_shared(); From 1a12c2c3ea218b3a7d4e4b7e7e86b1e64f0d9576 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Wed, 30 Apr 2025 16:34:08 -0700 Subject: [PATCH 04/13] More clang fixes. --- include/maddy/strongparser.h | 6 ++---- tests/maddy/test_maddy_emphasizedparser.cpp | 14 +++++++++----- 2 files changed, 11 insertions(+), 9 deletions(-) diff --git a/include/maddy/strongparser.h b/include/maddy/strongparser.h index 5cf9c38..8a66d6e 100644 --- a/include/maddy/strongparser.h +++ b/include/maddy/strongparser.h @@ -59,11 +59,9 @@ class StrongParser : public LineParser // }; static std::vector res{ std::regex{R"((?!.*`.*|.*.*)\*\*(?!.*`.*|.*<\/code>.*)" - "([^\*\*]*)\*\*(?!.*`.*|.*<\/code>.*))" - }, + "([^\*\*]*)\*\*(?!.*`.*|.*<\/code>.*))"}, std::regex{R"((?!.*`.*|.*.*)__(?!.*`.*|.*<\/code>.*)" - "([^__]*)__(?!.*`.*|.*<\/code>.*))" - } + "([^__]*)__(?!.*`.*|.*<\/code>.*))"} }; static std::string replacement = "$1"; for (const auto& re : res) diff --git a/tests/maddy/test_maddy_emphasizedparser.cpp b/tests/maddy/test_maddy_emphasizedparser.cpp index 8da0b4f..a70c248 100644 --- a/tests/maddy/test_maddy_emphasizedparser.cpp +++ b/tests/maddy/test_maddy_emphasizedparser.cpp @@ -67,8 +67,9 @@ TEST(MADDY_EMPHASIZEDPARSER, ItReplacesUnderscoresWithMultipleWords) TEST(MADDY_EMPHASIZEDPARSER, ItAllowsDoubleUnderscores) { - // I'm not sure if this is standard or not, but this is how the github markdown - // parser behaves. Other things I've seen want it to *not* match. + // I'm not sure if this is standard or not, but this is how the github + // markdown parser behaves. Other things I've seen want it to *not* + // match. std::string text = "some __text testing it_ out"; std::string expected = "some _text testing it out"; auto emphasizedParser = std::make_shared(); @@ -80,8 +81,10 @@ TEST(MADDY_EMPHASIZEDPARSER, ItAllowsDoubleUnderscores) TEST(MADDY_EMPHASIZEDPARSER, ItDoesntReplaceUnderscoresInsideCodeBlocks) { - std::string text = "Stuff inside blocks _shouldn't be emphasized_ at all"; - std::string expected = "Stuff inside blocks _shouldn't be emphasized_ at all"; + std::string text = + "Stuff inside blocks _shouldn't be emphasized_ at all"; + std::string expected = + "Stuff inside blocks _shouldn't be emphasized_ at all"; auto emphasizedParser = std::make_shared(); emphasizedParser->Parse(text); @@ -92,7 +95,8 @@ TEST(MADDY_EMPHASIZEDPARSER, ItDoesntReplaceUnderscoresInsideCodeBlocks) TEST(MADDY_EMPHASIZEDPARSER, ItDoesNotReplaceUnderscoresInURLs) { std::string text = "[Link Title](http://example.com/what_you_didn't_know)"; - std::string expected = "[Link Title](http://example.com/what_you_didn't_know)"; + std::string expected = + "[Link Title](http://example.com/what_you_didn't_know)"; auto emphasizedParser = std::make_shared(); emphasizedParser->Parse(text); From 8e2c44c337719b082cb445157d00737522bd7dc0 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Wed, 30 Apr 2025 16:36:31 -0700 Subject: [PATCH 05/13] A clang fix turned out to break stuff! --- include/maddy/strongparser.h | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/include/maddy/strongparser.h b/include/maddy/strongparser.h index 8a66d6e..a28d18b 100644 --- a/include/maddy/strongparser.h +++ b/include/maddy/strongparser.h @@ -58,10 +58,12 @@ class StrongParser : public LineParser // } // }; static std::vector res{ - std::regex{R"((?!.*`.*|.*.*)\*\*(?!.*`.*|.*<\/code>.*)" - "([^\*\*]*)\*\*(?!.*`.*|.*<\/code>.*))"}, - std::regex{R"((?!.*`.*|.*.*)__(?!.*`.*|.*<\/code>.*)" - "([^__]*)__(?!.*`.*|.*<\/code>.*))"} + std::regex{ + R"((?!.*`.*|.*.*)\*\*(?!.*`.*|.*<\/code>.*)([^\*\*]*)\*\*(?!.*`.*|.*<\/code>.*))" + }, + std::regex{ + R"((?!.*`.*|.*.*)__(?!.*`.*|.*<\/code>.*)([^__]*)__(?!.*`.*|.*<\/code>.*))" + } }; static std::string replacement = "$1"; for (const auto& re : res) From 30d6cf9d7ab3fd2e62acd1e18073f29dfe7008e4 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Wed, 30 Apr 2025 16:38:07 -0700 Subject: [PATCH 06/13] Fixed double negative phrasing. --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 25072a3..251add8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,7 +14,7 @@ maddy uses [semver versioning](https://semver.org/). ## Upcoming -* ![**FIXED**](https://img.shields.io/badge/-FIXED-%23090) Do not create emphasis tags not at word boundaries, i.e. `only_internal_underscores`. +* ![**FIXED**](https://img.shields.io/badge/-FIXED-%23090) Only create emphasis tags at word boundaries, i.e. `not only_internal_underscores`. * ... ## version 1.5.0 2025-04-21 From bfb0ff565a60002abad3fc318657dad2f05cb163 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Fri, 21 Aug 2026 13:04:36 -0700 Subject: [PATCH 07/13] Fix for emphasis and strong with internal and extra underscores. Addresses comments from ages ago (sorry!) and also does the correct behavior with unbalanced underscores. Made with help from Claude, but checked and updated a lot. --- CHANGELOG.md | 3 +- include/maddy/emphasizedparser.h | 9 ++-- include/maddy/strongparser.h | 43 ++++++------------ tests/maddy/test_maddy_emphasizedparser.cpp | 40 +++++++++++++++-- tests/maddy/test_maddy_strongparser.cpp | 49 +++++++++++++++++---- 5 files changed, 97 insertions(+), 47 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 251add8..e2f1fe8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,7 +14,8 @@ maddy uses [semver versioning](https://semver.org/). ## Upcoming -* ![**FIXED**](https://img.shields.io/badge/-FIXED-%23090) Only create emphasis tags at word boundaries, i.e. `not only_internal_underscores`. +* ![**FIXED**](https://img.shields.io/badge/-FIXED-%23090) Only create strong and emphasis tags at word boundaries, i.e. `not only_internal_underscores`. +* ![**FIXED**](https://img.shields.io/badge/-FIXED-%23090) Only create strong and emphasis tags at word boundaries for underscores, and correctly leave any leftover delimiters outside the tag on either side, however many there are, i.e. `___text__` becomes `_text` and `_text_______` becomes `text______`. * ... ## version 1.5.0 2025-04-21 diff --git a/include/maddy/emphasizedparser.h b/include/maddy/emphasizedparser.h index af30928..377e674 100644 --- a/include/maddy/emphasizedparser.h +++ b/include/maddy/emphasizedparser.h @@ -40,12 +40,13 @@ class EmphasizedParser : public LineParser */ void Parse(std::string& line) override { - // Modifed from previous version, with help from - // https://stackoverflow.com/questions/61346949/regex-for-markdown-emphasis + // The leading and trailing `(_*)` groups absorb any leftover underscores + // from an unbalanced run (e.g. `__foo_` or `_foo____`), re-emitted + // outside the tag instead of into its content. static std::regex re( - R"((?!.*`.*|.*.*)\b_(?![\s])(?!.*`.*|.*<\/code>.*)(.*?[^\s])_\b(?!.*`.*|.*<\/code>.*))" + R"((?!.*`.*|.*.*)\b(_*)_(?![\s_])(?!.*`.*|.*<\/code>.*)(.*?[^\s])_(_*)\b(?!.*`.*|.*<\/code>.*))" ); - static std::string replacement = "$1"; + static std::string replacement = "$1$2$3"; line = std::regex_replace(line, re, replacement); } diff --git a/include/maddy/strongparser.h b/include/maddy/strongparser.h index a28d18b..b6c0de7 100644 --- a/include/maddy/strongparser.h +++ b/include/maddy/strongparser.h @@ -40,36 +40,21 @@ class StrongParser : public LineParser */ void Parse(std::string& line) override { - // This version of the regex is changed exactly the same way - // that the regex for the emphasized parser was changed, and - // it then passes all the 'disabled' tests in the 'strong parser' - // test, but then it fails general parsing. For some reason, - // "__text__" translates "text" even though there - // are no word boundaries at the correct places. It's weird! - - // static std::vector res{ - // std::regex{ - // R"((?!.*`.*|.*.*)\b\*\*(?![\s])(?!.*`.*|.*<\/code>.*)" - // "(.*?[^\s])\*\*\b(?!.*`.*|.*<\/code>.*))" - // }, - // std::regex{ - // R"((?!.*`.*|.*.*)\b__(?![\s])(?!.*`.*|.*<\/code>.*)" - // "(.*?[^\s])__\b(?!.*`.*|.*<\/code>.*))" - // } - // }; - static std::vector res{ - std::regex{ - R"((?!.*`.*|.*.*)\*\*(?!.*`.*|.*<\/code>.*)([^\*\*]*)\*\*(?!.*`.*|.*<\/code>.*))" - }, - std::regex{ - R"((?!.*`.*|.*.*)__(?!.*`.*|.*<\/code>.*)([^__]*)__(?!.*`.*|.*<\/code>.*))" - } + // `*` is not a word character, so `\b` next to it does not mean + // "edge of a delimiter run" the way it does for `_`; the asterisk + // variant is left without a word-boundary anchor. + static std::regex reAsterisk{ + R"((?!.*`.*|.*.*)\*\*(?![\s])(?!.*`.*|.*<\/code>.*)(.*?[^\s])\*\*(?!.*`.*|.*<\/code>.*))" + }; + // The leading and trailing `(_*)` groups absorb any leftover underscores + // from an unbalanced run on either side (e.g. `___text__` or + // `__text_______`), re-emitted outside the tag by the caller + // instead of being swallowed into its content. + static std::regex reUnderscore{ + R"((?!.*`.*|.*.*)\b(_*)__(?![\s_])(?!.*`.*|.*<\/code>.*)(.*?[^\s])__(_*)\b(?!.*`.*|.*<\/code>.*))" }; - static std::string replacement = "$1"; - for (const auto& re : res) - { - line = std::regex_replace(line, re, replacement); - } + line = std::regex_replace(line, reAsterisk, "$1"); + line = std::regex_replace(line, reUnderscore, "$1$2$3"); } }; // class StrongParser diff --git a/tests/maddy/test_maddy_emphasizedparser.cpp b/tests/maddy/test_maddy_emphasizedparser.cpp index a70c248..9e1cd52 100644 --- a/tests/maddy/test_maddy_emphasizedparser.cpp +++ b/tests/maddy/test_maddy_emphasizedparser.cpp @@ -67,11 +67,43 @@ TEST(MADDY_EMPHASIZEDPARSER, ItReplacesUnderscoresWithMultipleWords) TEST(MADDY_EMPHASIZEDPARSER, ItAllowsDoubleUnderscores) { - // I'm not sure if this is standard or not, but this is how the github - // markdown parser behaves. Other things I've seen want it to *not* - // match. + // Per CommonMark, a leftover delimiter from an unbalanced run renders + // outside the tag it didn't pair into, not inside it. std::string text = "some __text testing it_ out"; - std::string expected = "some _text testing it out"; + std::string expected = "some _text testing it out"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItAllowsTrailingDoubleUnderscores) +{ + std::string text = "some _text testing it__ out"; + std::string expected = "some text testing it_ out"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItAllowsManyLeadingLeftoverUnderscores) +{ + std::string text = "some ____text testing it_ out"; + std::string expected = "some ___text testing it out"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItAllowsManyTrailingLeftoverUnderscores) +{ + std::string text = "some _text testing it____ out"; + std::string expected = "some text testing it___ out"; auto emphasizedParser = std::make_shared(); emphasizedParser->Parse(text); diff --git a/tests/maddy/test_maddy_strongparser.cpp b/tests/maddy/test_maddy_strongparser.cpp index 466068d..1211b0e 100644 --- a/tests/maddy/test_maddy_strongparser.cpp +++ b/tests/maddy/test_maddy_strongparser.cpp @@ -95,7 +95,7 @@ TEST(MADDY_STRONGPARSER, ItReplacesUnderscoresAtStringEdges) ASSERT_EQ(expected, text); } -TEST(DISABLED_MADDY_STRONGPARSER, ItDoesNotReplaceMarkdownWithInlineUnderscores) +TEST(MADDY_STRONGPARSER, ItDoesNotReplaceMarkdownWithInlineUnderscores) { std::string text = "some text__bla__text testing __it__ out"; std::string expected = "some text__bla__text testing it out"; @@ -106,7 +106,7 @@ TEST(DISABLED_MADDY_STRONGPARSER, ItDoesNotReplaceMarkdownWithInlineUnderscores) ASSERT_EQ(expected, text); } -TEST(DISABLED_MADDY_STRONGPARSER, ItOnlyReplacesUnderscoresAtWordBreaks) +TEST(MADDY_STRONGPARSER, ItOnlyReplacesUnderscoresAtWordBreaks) { std::string text = "some __text__bla__ testing __it__ out"; std::string expected = @@ -129,14 +129,45 @@ TEST(MADDY_STRONGPARSER, ItReplacesUnderscoresWithMultipleWords) ASSERT_EQ(expected, text); } -TEST(DISABLED_MADDY_STRONGPARSER, ItAllowsTripleUnderscores) +TEST(MADDY_STRONGPARSER, ItAllowsTripleUnderscores) { - // I'm not sure if this is standard or not, but this is how the github - // markdown parser behaves. Other things I've seen want it to *not* - // match. - + // Per CommonMark, a leftover delimiter from an unbalanced run renders + // outside the tag it didn't pair into, not inside it. std::string text = "some ___text testing it__ out"; - std::string expected = "some _text testing it out"; + std::string expected = "some _text testing it out"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_STRONGPARSER, ItAllowsTrailingTripleUnderscores) +{ + std::string text = "some __text testing it___ out"; + std::string expected = "some text testing it_ out"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_STRONGPARSER, ItAllowsManyLeadingLeftoverUnderscores) +{ + std::string text = "some ________text testing it__ out"; + std::string expected = "some ______text testing it out"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_STRONGPARSER, ItAllowsManyTrailingLeftoverUnderscores) +{ + std::string text = "some __text testing it_______ out"; + std::string expected = "some text testing it_____ out"; auto strongParser = std::make_shared(); strongParser->Parse(text); @@ -157,7 +188,7 @@ TEST(MADDY_STRONGPARSER, ItDoesntReplaceUnderscoresInsideCodeBlocks) ASSERT_EQ(expected, text); } -TEST(DISABLED_MADDY_STRONGPARSER, ItDoesNotReplaceUnderscoresInURLs) +TEST(MADDY_STRONGPARSER, ItDoesNotReplaceUnderscoresInURLs) { std::string text = "[Link Title](http://example.com/what__you__didn't__know)"; std::string expected = From 57f04f89a7b5d77150bcad1cada831a182a5ed0a Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Fri, 21 Aug 2026 13:11:37 -0700 Subject: [PATCH 08/13] Update CMake version so that github can find VS 2026. (2022 was removed) --- .github/workflows/run-tests.yml | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/run-tests.yml b/.github/workflows/run-tests.yml index b118bec..4a415b8 100644 --- a/.github/workflows/run-tests.yml +++ b/.github/workflows/run-tests.yml @@ -12,7 +12,7 @@ jobs: - uses: actions/checkout@v4 - uses: lukka/get-cmake@latest with: - cmakeVersion: "~3.25.0" # <--= optional, use most recent 3.25.x version + cmakeVersion: "~4.2.0" # <--= optional, use most recent 4.2.x version ninjaVersion: "^1.11.1" # <--= optional, use most recent 1.x version - name: build run: | @@ -30,13 +30,13 @@ jobs: - uses: actions/checkout@v4 - uses: lukka/get-cmake@latest with: - cmakeVersion: "~3.25.0" # <--= optional, use most recent 3.25.x version + cmakeVersion: "~4.2.0" # <--= optional, use most recent 4.2.x version ninjaVersion: "^1.11.1" # <--= optional, use most recent 1.x version - name: build run: | mkdir tmp cd tmp - cmake -G "Visual Studio 17 2022" -A x64 -DMADDY_BUILD_WITH_TESTS=ON .. + cmake -G "Visual Studio 18 2026" -A x64 -DMADDY_BUILD_WITH_TESTS=ON .. cmake --build . --config Debug - name: run tests run: | @@ -48,7 +48,7 @@ jobs: - uses: actions/checkout@v4 - uses: lukka/get-cmake@latest with: - cmakeVersion: "~3.25.0" # <--= optional, use most recent 3.25.x version + cmakeVersion: "~4.2.0" # <--= optional, use most recent 4.2.x version ninjaVersion: "^1.11.1" # <--= optional, use most recent 1.x version - name: build run: | From 96c354438a66e1d1526ac40ebe73d77598c72691 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Fri, 21 Aug 2026 13:15:02 -0700 Subject: [PATCH 09/13] Fix changelog wording. --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index e2f1fe8..2614f9f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -15,7 +15,7 @@ maddy uses [semver versioning](https://semver.org/). ## Upcoming * ![**FIXED**](https://img.shields.io/badge/-FIXED-%23090) Only create strong and emphasis tags at word boundaries, i.e. `not only_internal_underscores`. -* ![**FIXED**](https://img.shields.io/badge/-FIXED-%23090) Only create strong and emphasis tags at word boundaries for underscores, and correctly leave any leftover delimiters outside the tag on either side, however many there are, i.e. `___text__` becomes `_text` and `_text_______` becomes `text______`. +* ![**FIXED**](https://img.shields.io/badge/-FIXED-%23090) Correctly leave any leftover strong or emphasis delimiters outside the tag on either side, however many there are, i.e. `___text__` becomes `_text` and `_text_______` becomes `text______`. * ... ## version 1.5.0 2025-04-21 From 4651fc98ceb55bebc89cd4c79c34ef6b092e4ee3 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Fri, 21 Aug 2026 15:43:41 -0700 Subject: [PATCH 10/13] Split tables parsing into Maddy-style or CommonMark-style. Maddy's parsing of tables is a one-off, and works for both the body and the footer of tables. CommonMark doesn't support table footers at all. However, if you're using maddy as part of a system that needs to round-trip HTML through MD and back again, the table format that a CommonMark-style exporter creates won't be parsed by Maddy. So, this update splits table parsing into two forms: the form Maddy invented, and the form CommonMark/GFM expects. This can be set using a new MADDY_SPECIFIC_PARSER flag, and by default, it parses its own invented format. When turned off, the parser instead parses GFM-style tables. All of this code is just Claude, though all the design was me. --- .github/workflows/run-tests.yml | 14 +- CHANGELOG.md | 1 + docs/definitions.md | 47 ++++ include/maddy/parser.h | 11 +- include/maddy/parserconfig.h | 9 +- include/maddy/tableparser.h | 320 +++++++++++++++++++++---- tests/maddy/test_maddy_parser.cpp | 43 ++++ tests/maddy/test_maddy_tableparser.cpp | 106 ++++++++ 8 files changed, 496 insertions(+), 55 deletions(-) diff --git a/.github/workflows/run-tests.yml b/.github/workflows/run-tests.yml index 29cf5d6..144e73c 100644 --- a/.github/workflows/run-tests.yml +++ b/.github/workflows/run-tests.yml @@ -12,8 +12,8 @@ jobs: - uses: actions/checkout@v6 - uses: lukka/get-cmake@latest with: - cmakeVersion: "~3.25.0" - ninjaVersion: "^1.11.1" + cmakeVersion: "~4.2.0" # <--= optional, use most recent 4.2.x version + ninjaVersion: "^1.11.1" # <--= optional, use most recent 1.x version - name: build run: | mkdir tmp @@ -39,13 +39,13 @@ jobs: - uses: actions/checkout@v6 - uses: lukka/get-cmake@latest with: - cmakeVersion: "~3.25.0" - ninjaVersion: "^1.11.1" + cmakeVersion: "~4.2.0" # <--= optional, use most recent 4.2.x version + ninjaVersion: "^1.11.1" # <--= optional, use most recent 1.x version - name: build run: | mkdir tmp cd tmp - cmake -G "Visual Studio 17 2022" -A x64 -DMADDY_BUILD_WITH_TESTS=ON .. + cmake -G "Visual Studio 18 2026" -A x64 -DMADDY_BUILD_WITH_TESTS=ON .. cmake --build . --config Debug - name: run tests run: | @@ -57,8 +57,8 @@ jobs: - uses: actions/checkout@v6 - uses: lukka/get-cmake@latest with: - cmakeVersion: "~3.25.0" - ninjaVersion: "^1.11.1" + cmakeVersion: "~4.2.0" # <--= optional, use most recent 4.2.x version + ninjaVersion: "^1.11.1" # <--= optional, use most recent 1.x version - name: build run: | mkdir tmp diff --git a/CHANGELOG.md b/CHANGELOG.md index a16cc40..c6f08e9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,7 @@ maddy uses [semver versioning](https://semver.org/). ## Upcoming +* ![**ADDED**](https://img.shields.io/badge/-ADDED-%23099) New `maddy::types::MADDY_SPECIFIC_PARSER` config flag (on by default, keeping current behavior). Turning it off switches `TableParser` from maddy's own `|table>` sigil syntax to standard GitHub-flavored-Markdown pipe tables (which have no footer concept). * ... ## version 1.6.0 2025-07-26 diff --git a/docs/definitions.md b/docs/definitions.md index 1a3800e..81d0e79 100644 --- a/docs/definitions.md +++ b/docs/definitions.md @@ -385,6 +385,53 @@ becomes ``` table header and footer are optional +This is maddy's own historical table syntax, and it's used by default +(`maddy::types::MADDY_SPECIFIC_PARSER`). If you need standard +GitHub-flavored-Markdown pipe tables instead, disable it in config: + +```cpp +std::shared_ptr config = std::make_shared(); +config->enabledParsers &= ~maddy::types::MADDY_SPECIFIC_PARSER; + +std::shared_ptr parser = std::make_shared(config); +std::string htmlOutput = parser->Parse(markdownInput); +``` + +After this, tables look like: + +``` +| Left header | middle header | last header | +| --- | --- | --- | +| cell 1 | cell 2 | cell 3 | +| cell 4 | cell 5 | cell 6 | +``` +becomes +```html + + + + + + + + + + + + + + + + + + + + +
Left headermiddle headerlast header
cell 1cell 2cell 3
cell 4cell 5cell 6
+``` +GFM pipe tables have no footer concept, so this mode never produces a +``. + ## LaTeX(MathJax) block support To turn on the LaTeX support - which basically is only a diff --git a/include/maddy/parser.h b/include/maddy/parser.h index 660752f..bcdf560 100644 --- a/include/maddy/parser.h +++ b/include/maddy/parser.h @@ -278,10 +278,17 @@ class Parser } else if ((!this->config || (this->config->enabledParsers & maddy::types::TABLE_PARSER) != 0) && - maddy::TableParser::IsStartingLine(line)) + maddy::TableParser::IsStartingLine( + line, + !this->config || (this->config->enabledParsers & + maddy::types::MADDY_SPECIFIC_PARSER) != 0 + )) { parser = std::make_shared( - [this](std::string& line) { this->runLineParser(line); }, nullptr + [this](std::string& line) { this->runLineParser(line); }, + nullptr, + !this->config || (this->config->enabledParsers & + maddy::types::MADDY_SPECIFIC_PARSER) != 0 ); } else if ((!this->config || (this->config->enabledParsers & diff --git a/include/maddy/parserconfig.h b/include/maddy/parserconfig.h index f95ee04..74ab2c9 100644 --- a/include/maddy/parserconfig.h +++ b/include/maddy/parserconfig.h @@ -44,8 +44,13 @@ enum PARSER_TYPE : uint32_t UNORDERED_LIST_PARSER = 0b100000000000000000, LATEX_BLOCK_PARSER = 0b1000000000000000000, - DEFAULT = 0b0111111111110111111, - ALL = 0b1111111111111111111, + // Not a parser of its own: gates maddy's own historical markdown dialect + // wherever a parser supports both that and a more standard alternative + // (currently just TableParser's `|table>` syntax vs. GFM pipe tables). + MADDY_SPECIFIC_PARSER = 0b10000000000000000000, + + DEFAULT = 0b10111111111110111111, + ALL = 0b11111111111111111111, }; // clang-format on diff --git a/include/maddy/tableparser.h b/include/maddy/tableparser.h index 9f1051f..1e6ec84 100644 --- a/include/maddy/tableparser.h +++ b/include/maddy/tableparser.h @@ -7,10 +7,14 @@ // ----------------------------------------------------------------------------- #include +#include #include +#include #include +#include #include "maddy/blockparser.h" +#include "maddy/paragraphparser.h" // ----------------------------------------------------------------------------- @@ -21,7 +25,37 @@ namespace maddy { /** * TableParser * - * For more information, see the docs folder. + * Supports two independent syntaxes, chosen with `useMaddySpecificMarkdown` + * (see `maddy::types::MADDY_SPECIFIC_PARSER`). + * + * When true (the default, and maddy's original behavior), a table uses + * maddy's own sigils: + * + * ``` + * |table> + * Left header|middle header|last header + * - | - | - + * Cell A1|Cell B1|Cell C1 + * - | - | - + * Foot A|Foot B|Foot C + * |`. Since `IsStartingLine` only sees one line at a time in this + * mode, it can't yet tell a table header from an ordinary line that happens + * to contain a `|`; if the following line isn't a valid separator row, both + * lines are handed off to a ParagraphParser instead. * * @class */ @@ -35,32 +69,48 @@ class TableParser : public BlockParser * @param {std::function} parseLineCallback * @param {std::function(const std::string& * line)>} getBlockParserForLineCallback + * @param {bool} useMaddySpecificMarkdown */ TableParser( std::function parseLineCallback, std::function(const std::string& line)> - getBlockParserForLineCallback + getBlockParserForLineCallback, + bool useMaddySpecificMarkdown = true ) : BlockParser(parseLineCallback, getBlockParserForLineCallback) + , useMaddySpecificMarkdown(useMaddySpecificMarkdown) , isStarted(false) , isFinished(false) , currentBlock(0) , currentRow(0) + , gfmState(GfmState::EXPECT_HEADER) {} /** * IsStartingLine * - * If the line has exact `|table>`, then it is starting the table. + * With maddy-specific markdown, a table starts with exact `|table>`. + * With GFM markdown, a table can only start with a row that has at least + * one `|` separating two cells; whether it really is a table is only + * known once the following line (the separator row) has been seen. * * @method * @param {const std::string&} line + * @param {bool} useMaddySpecificMarkdown * @return {bool} */ - static bool IsStartingLine(const std::string& line) + static bool IsStartingLine( + const std::string& line, + bool useMaddySpecificMarkdown = true + ) { - static std::string matchString("|table>"); - return line == matchString; + if (useMaddySpecificMarkdown) + { + static std::string matchString("|table>"); + return line == matchString; + } + + return IsTableRow(line); } /** @@ -74,52 +124,21 @@ class TableParser : public BlockParser */ void AddLine(std::string& line) override { - if (!this->isStarted && line == "|table>") + if (this->useMaddySpecificMarkdown) { - this->isStarted = true; - return; + this->AddLineMaddyStyle(line); } - - if (this->isStarted) + else { - if (line == "- | - | -") - { - ++this->currentBlock; - this->currentRow = 0; - return; - } - - if (line == "|parseBlock(emptyLine); - this->isFinished = true; - return; - } - - if (this->table.size() < this->currentBlock + 1) - { - this->table.push_back(std::vector>()); - } - this->table[this->currentBlock].push_back(std::vector()); - - std::string segment; - std::stringstream streamToSplit(line); - - while (std::getline(streamToSplit, segment, '|')) - { - this->parseLine(segment); - this->table[this->currentBlock][this->currentRow].push_back(segment); - } - - ++this->currentRow; + this->AddLineGfm(line); } } /** * IsFinished * - * A table ends with `|"; @@ -221,11 +242,222 @@ class TableParser : public BlockParser } private: + bool useMaddySpecificMarkdown; + + // --- maddy-specific-markdown mode state --- bool isStarted; bool isFinished; uint32_t currentBlock; uint32_t currentRow; std::vector>> table; + + void AddLineMaddyStyle(std::string& line) + { + if (!this->isStarted && line == "|table>") + { + this->isStarted = true; + return; + } + + if (this->isStarted) + { + if (line == "- | - | -") + { + ++this->currentBlock; + this->currentRow = 0; + return; + } + + if (line == "|parseBlock(emptyLine); + this->isFinished = true; + return; + } + + if (this->table.size() < this->currentBlock + 1) + { + this->table.push_back(std::vector>()); + } + this->table[this->currentBlock].push_back(std::vector()); + + std::string segment; + std::stringstream streamToSplit(line); + + while (std::getline(streamToSplit, segment, '|')) + { + this->parseLine(segment); + this->table[this->currentBlock][this->currentRow].push_back(segment); + } + + ++this->currentRow; + } + } + + // --- GFM-pipe-table mode state --- + enum class GfmState { EXPECT_HEADER, EXPECT_SEPARATOR, IN_BODY }; + + GfmState gfmState; + std::string headerLine; + std::shared_ptr fallbackParser; + + void AddLineGfm(std::string& line) + { + if (this->fallbackParser) + { + this->fallbackParser->AddLine(line); + + if (this->fallbackParser->IsFinished()) + { + this->result << this->fallbackParser->GetResult().str(); + this->isFinished = true; + } + + return; + } + + switch (this->gfmState) + { + case GfmState::EXPECT_HEADER: + this->headerLine = line; + this->gfmState = GfmState::EXPECT_SEPARATOR; + return; + + case GfmState::EXPECT_SEPARATOR: + if ( + IsSeparatorRow(line) && + SplitRow(line).size() == SplitRow(this->headerLine).size() + ) + { + this->WriteGfmHeader(); + this->gfmState = GfmState::IN_BODY; + } + else + { + this->FallBackToParagraph(line); + } + return; + + case GfmState::IN_BODY: + if (line.empty()) + { + this->result << "
"; + this->isFinished = true; + } + else + { + this->WriteGfmRow(line); + } + return; + } + } + + static bool IsTableRow(const std::string& line) + { + return line.find('|') != std::string::npos && + line.find_first_not_of(" \t") != std::string::npos; + } + + static bool IsSeparatorRow(const std::string& line) + { + if (!IsTableRow(line)) + { + return false; + } + + static const std::regex cellRe("^:?-+:?$"); + + for (const std::string& cell : SplitRow(line)) + { + if (!std::regex_match(cell, cellRe)) + { + return false; + } + } + + return true; + } + + static std::vector SplitRow(const std::string& line) + { + std::vector cells; + std::stringstream stream(line); + std::string cell; + + while (std::getline(stream, cell, '|')) + { + Trim(cell); + + if (!cell.empty()) + { + cells.push_back(cell); + } + } + + return cells; + } + + static void Trim(std::string& str) + { + size_t first = str.find_first_not_of(" \t"); + + if (first == std::string::npos) + { + str.clear(); + return; + } + + size_t last = str.find_last_not_of(" \t"); + str = str.substr(first, last - first + 1); + } + + void WriteGfmHeader() + { + this->result << ""; + + for (std::string cell : SplitRow(this->headerLine)) + { + this->parseLine(cell); + this->result << ""; + } + + this->result << ""; + } + + void WriteGfmRow(const std::string& line) + { + this->result << ""; + + for (std::string cell : SplitRow(line)) + { + this->parseLine(cell); + this->result << ""; + } + + this->result << ""; + } + + void FallBackToParagraph(const std::string& secondLine) + { + this->fallbackParser = std::make_shared( + [this](std::string& l) { this->parseLine(l); }, + nullptr, + true + ); + + std::string first = this->headerLine; + this->fallbackParser->AddLine(first); + + std::string second = secondLine; + this->fallbackParser->AddLine(second); + + if (this->fallbackParser->IsFinished()) + { + this->result << this->fallbackParser->GetResult().str(); + this->isFinished = true; + } + } }; // class TableParser // ----------------------------------------------------------------------------- diff --git a/tests/maddy/test_maddy_parser.cpp b/tests/maddy/test_maddy_parser.cpp index ed8530a..83a20cb 100644 --- a/tests/maddy/test_maddy_parser.cpp +++ b/tests/maddy/test_maddy_parser.cpp @@ -82,3 +82,46 @@ TEST(MADDY_PARSER, ItShouldNotParseInlineCodeInHeadlineIfDisabled) ASSERT_EQ(expectedHTML, output); } + +TEST(MADDY_PARSER, ItShouldParseGfmTablesWhenMaddySpecificParserIsDisabled) +{ + const std::string tableTest = + "| Left header | middle header | last header |\n" + "| --- | --- | --- |\n" + "| cell 1 | cell 2 | cell 3 |\n" + "| cell 4 | cell 5 | cell 6 |\n"; + const std::string expectedHTML = + "
" << cell << "
" << cell << "
Left headermiddle headerlast " + "header
cell 1cell 2cell " + "3
cell 4cell 5cell " + "6
"; + std::stringstream markdown(tableTest); + auto config = std::make_shared(); + config->enabledParsers &= ~maddy::types::MADDY_SPECIFIC_PARSER; + auto parser = std::make_shared(config); + + const std::string output = parser->Parse(markdown); + + ASSERT_EQ(expectedHTML, output); +} + +TEST( + MADDY_PARSER, + ItShouldNotParseMaddySpecificTableSyntaxWhenMaddySpecificParserIsDisabled +) +{ + const std::string tableTest = + "|table>\n" + "A|B\n" + "- | - | -\n" + "1|2\n" + "|(); + config->enabledParsers &= ~maddy::types::MADDY_SPECIFIC_PARSER; + auto parser = std::make_shared(config); + + const std::string output = parser->Parse(markdown); + + ASSERT_EQ(std::string::npos, output.find("")); +} diff --git a/tests/maddy/test_maddy_tableparser.cpp b/tests/maddy/test_maddy_tableparser.cpp index 5d50d92..4bfbe66 100644 --- a/tests/maddy/test_maddy_tableparser.cpp +++ b/tests/maddy/test_maddy_tableparser.cpp @@ -67,3 +67,109 @@ TEST_F(MADDY_TABLEPARSER, ItReplacesMarkdownWithAnHtmlTable) ASSERT_EQ(expected, outputString); } + +// ----------------------------------------------------------------------------- +// GFM pipe-table mode (useMaddySpecificMarkdown = false) +// ----------------------------------------------------------------------------- + +class MADDY_TABLEPARSER_GFM : public ::testing::Test +{ +protected: + std::shared_ptr tableParser; + + void SetUp() override + { + this->tableParser = + std::make_shared(nullptr, nullptr, false); + } +}; + +TEST_F( + MADDY_TABLEPARSER_GFM, + IsStartingLineReturnsTrueForAnyLineWithAPipe +) +{ + ASSERT_TRUE(maddy::TableParser::IsStartingLine("| a | b |", false)); +} + +TEST_F( + MADDY_TABLEPARSER_GFM, + IsStartingLineReturnsFalseForABlankLine +) +{ + ASSERT_FALSE(maddy::TableParser::IsStartingLine(" ", false)); +} + +TEST_F(MADDY_TABLEPARSER_GFM, IsFinishedReturnsFalseInTheBeginning) +{ + ASSERT_FALSE(tableParser->IsFinished()); +} + +TEST_F(MADDY_TABLEPARSER_GFM, ItReplacesMarkdownWithAnHtmlTableAndHasNoFooter) +{ + std::vector markdown = { + "| Left header | middle header | last header |", + "| --- | --- | --- |", + "| cell 1 | cell 2 | cell 3 |", + "| cell 4 | cell 5 | cell 6 |", + "" + }; + std::string expected = + "
Left headermiddle headerlast " + "header
cell 1cell 2cell " + "3
cell 4cell 5cell " + "6
"; + + for (std::string md : markdown) + { + tableParser->AddLine(md); + } + + ASSERT_TRUE(tableParser->IsFinished()); + ASSERT_EQ(expected, tableParser->GetResult().str()); +} + +TEST_F( + MADDY_TABLEPARSER_GFM, + ItFallsBackToAParagraphWhenTheSecondLineIsNotASeparatorRow +) +{ + std::string first = "not | a | table"; + std::string second = "just some more text"; + std::string third = ""; + std::string expected = "

not | a | table just some more text

"; + + tableParser->AddLine(first); + tableParser->AddLine(second); + + if (!tableParser->IsFinished()) + { + tableParser->AddLine(third); + } + + ASSERT_TRUE(tableParser->IsFinished()); + ASSERT_EQ(expected, tableParser->GetResult().str()); +} + +TEST_F( + MADDY_TABLEPARSER_GFM, + OldMaddySpecificSyntaxIsNotRecognizedAsATable +) +{ + std::string first = "|table>"; + std::string second = "Left header|middle header|last header"; + std::string third = ""; + + tableParser->AddLine(first); + tableParser->AddLine(second); + + if (!tableParser->IsFinished()) + { + tableParser->AddLine(third); + } + + ASSERT_TRUE(tableParser->IsFinished()); + ASSERT_EQ( + std::string::npos, tableParser->GetResult().str().find("") + ); +} From 5f4b5effec90926ff3af313f4944972c1b4c2fe7 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Fri, 21 Aug 2026 15:52:09 -0700 Subject: [PATCH 11/13] Format for clang. --- include/maddy/parser.h | 6 +++--- include/maddy/tableparser.h | 20 ++++++++++---------- tests/maddy/test_maddy_tableparser.cpp | 19 ++++--------------- 3 files changed, 17 insertions(+), 28 deletions(-) diff --git a/include/maddy/parser.h b/include/maddy/parser.h index bcdf560..a3c8a18 100644 --- a/include/maddy/parser.h +++ b/include/maddy/parser.h @@ -59,7 +59,7 @@ class Parser */ static const std::string& version() { - static const std::string v = "1.6.0"; // MADDY_VERSION_LINE_REPLACEMENT + static const std::string v = "1.5.0"; return v; } @@ -281,14 +281,14 @@ class Parser maddy::TableParser::IsStartingLine( line, !this->config || (this->config->enabledParsers & - maddy::types::MADDY_SPECIFIC_PARSER) != 0 + maddy::types::MADDY_SPECIFIC_PARSER) != 0 )) { parser = std::make_shared( [this](std::string& line) { this->runLineParser(line); }, nullptr, !this->config || (this->config->enabledParsers & - maddy::types::MADDY_SPECIFIC_PARSER) != 0 + maddy::types::MADDY_SPECIFIC_PARSER) != 0 ); } else if ((!this->config || (this->config->enabledParsers & diff --git a/include/maddy/tableparser.h b/include/maddy/tableparser.h index 1e6ec84..7f7e6bc 100644 --- a/include/maddy/tableparser.h +++ b/include/maddy/tableparser.h @@ -100,8 +100,7 @@ class TableParser : public BlockParser * @return {bool} */ static bool IsStartingLine( - const std::string& line, - bool useMaddySpecificMarkdown = true + const std::string& line, bool useMaddySpecificMarkdown = true ) { if (useMaddySpecificMarkdown) @@ -296,7 +295,12 @@ class TableParser : public BlockParser } // --- GFM-pipe-table mode state --- - enum class GfmState { EXPECT_HEADER, EXPECT_SEPARATOR, IN_BODY }; + enum class GfmState + { + EXPECT_HEADER, + EXPECT_SEPARATOR, + IN_BODY + }; GfmState gfmState; std::string headerLine; @@ -325,10 +329,8 @@ class TableParser : public BlockParser return; case GfmState::EXPECT_SEPARATOR: - if ( - IsSeparatorRow(line) && - SplitRow(line).size() == SplitRow(this->headerLine).size() - ) + if (IsSeparatorRow(line) && + SplitRow(line).size() == SplitRow(this->headerLine).size()) { this->WriteGfmHeader(); this->gfmState = GfmState::IN_BODY; @@ -441,9 +443,7 @@ class TableParser : public BlockParser void FallBackToParagraph(const std::string& secondLine) { this->fallbackParser = std::make_shared( - [this](std::string& l) { this->parseLine(l); }, - nullptr, - true + [this](std::string& l) { this->parseLine(l); }, nullptr, true ); std::string first = this->headerLine; diff --git a/tests/maddy/test_maddy_tableparser.cpp b/tests/maddy/test_maddy_tableparser.cpp index 4bfbe66..592fd7e 100644 --- a/tests/maddy/test_maddy_tableparser.cpp +++ b/tests/maddy/test_maddy_tableparser.cpp @@ -84,18 +84,12 @@ class MADDY_TABLEPARSER_GFM : public ::testing::Test } }; -TEST_F( - MADDY_TABLEPARSER_GFM, - IsStartingLineReturnsTrueForAnyLineWithAPipe -) +TEST_F(MADDY_TABLEPARSER_GFM, IsStartingLineReturnsTrueForAnyLineWithAPipe) { ASSERT_TRUE(maddy::TableParser::IsStartingLine("| a | b |", false)); } -TEST_F( - MADDY_TABLEPARSER_GFM, - IsStartingLineReturnsFalseForABlankLine -) +TEST_F(MADDY_TABLEPARSER_GFM, IsStartingLineReturnsFalseForABlankLine) { ASSERT_FALSE(maddy::TableParser::IsStartingLine(" ", false)); } @@ -151,10 +145,7 @@ TEST_F( ASSERT_EQ(expected, tableParser->GetResult().str()); } -TEST_F( - MADDY_TABLEPARSER_GFM, - OldMaddySpecificSyntaxIsNotRecognizedAsATable -) +TEST_F(MADDY_TABLEPARSER_GFM, OldMaddySpecificSyntaxIsNotRecognizedAsATable) { std::string first = "|table>"; std::string second = "Left header|middle header|last header"; @@ -169,7 +160,5 @@ TEST_F( } ASSERT_TRUE(tableParser->IsFinished()); - ASSERT_EQ( - std::string::npos, tableParser->GetResult().str().find("
") - ); + ASSERT_EQ(std::string::npos, tableParser->GetResult().str().find("
")); } From 038e4f50d8337b910d7132b0d54c5b0f9f361e13 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Fri, 18 Sep 2026 08:15:22 -0700 Subject: [PATCH 12/13] Make regexes less complex. VS2026 is (currently) more strict than VS2022 in its regexes; I've started to hit regex_error(error_complexity) throws that I didn't get before. There's an issue out that might fix this (https://github.com/microsoft/STL/issues/6452) but in the meantime, these adjustments to the code blocks and the regexes make things simpler so they don't hit the complexity limits of VS2026 for the inputs I was hitting. At the same time, some edge cases in the original (with emphasis plus backticks) were incorrect according to the markdown spec. Those are now spec-compliant. The majority of this code comes from Claude, so it's worth a stricter code review. --- include/maddy/codespanutils.h | 131 ++++++++++++++++++ include/maddy/emphasizedparser.h | 9 +- include/maddy/italicparser.h | 10 +- include/maddy/strikethroughparser.h | 9 +- include/maddy/strongparser.h | 16 +-- tests/CMakeLists.txt | 2 +- tests/maddy/test_maddy_emphasizedparser.cpp | 35 +++++ tests/maddy/test_maddy_italicparser.cpp | 35 +++++ .../maddy/test_maddy_strikethroughparser.cpp | 35 +++++ tests/maddy/test_maddy_strongparser.cpp | 39 +++++- 10 files changed, 299 insertions(+), 22 deletions(-) create mode 100644 include/maddy/codespanutils.h diff --git a/include/maddy/codespanutils.h b/include/maddy/codespanutils.h new file mode 100644 index 0000000..4e33a04 --- /dev/null +++ b/include/maddy/codespanutils.h @@ -0,0 +1,131 @@ +/* + * This project is licensed under the MIT license. For more information see the + * LICENSE file. + */ +#pragma once + +// ----------------------------------------------------------------------------- + +#include +#include +#include +#include +#include + +// ----------------------------------------------------------------------------- + +namespace maddy { + +// ----------------------------------------------------------------------------- + +/** + * FindProtectedSpans + * + * Finds non-overlapping [start, end) ranges of a line that inline + * delimiter parsers (strong, emphasized, strikethrough, italic) must not + * alter: a `...` HTML span, or a code span delimited by a + * run of N backtick characters and a later run of exactly N backticks + * (per CommonMark, a run with no matching close is ordinary text, and + * scanning resumes right after it). + * + * @method + * @param {const std::string&} line + * @return {std::vector>} + */ +inline std::vector> FindProtectedSpans( + const std::string& line +) +{ + static const std::string openCodeTag = ""; + static const std::string closeCodeTag = ""; + + std::vector> spans; + std::size_t i = 0; + + while (i < line.size()) + { + if (line.compare(i, openCodeTag.size(), openCodeTag) == 0) + { + std::size_t closeStart = line.find(closeCodeTag, i + openCodeTag.size()); + if (closeStart != std::string::npos) + { + std::size_t end = closeStart + closeCodeTag.size(); + spans.emplace_back(i, end); + i = end; + continue; + } + } + + if (line[i] == '`') + { + std::size_t runStart = i; + while (i < line.size() && line[i] == '`') { ++i; } + std::size_t runLength = i - runStart; + + std::size_t searchPos = i; + while (searchPos < line.size()) + { + std::size_t closeStart = line.find('`', searchPos); + if (closeStart == std::string::npos) { break; } + + std::size_t closeEnd = closeStart; + while (closeEnd < line.size() && line[closeEnd] == '`') { ++closeEnd; } + + if (closeEnd - closeStart == runLength) + { + spans.emplace_back(runStart, closeEnd); + i = closeEnd; + break; + } + + searchPos = closeEnd; + } + + continue; + } + + ++i; + } + + return spans; +} + +/** + * ApplyOutsideProtectedSpans + * + * Runs `transform` on each stretch of `line` that falls outside its + * protected spans (see FindProtectedSpans), leaving the spans themselves + * untouched, then writes the reassembled result back into `line`. + * + * @method + * @param {std::string&} line + * @param {const std::function&} transform + * @return {void} + */ +inline void ApplyOutsideProtectedSpans( + std::string& line, + const std::function& transform +) +{ + std::string result; + std::size_t pos = 0; + + for (const auto& span : FindProtectedSpans(line)) + { + std::string segment = line.substr(pos, span.first - pos); + transform(segment); + result += segment; + result += line.substr(span.first, span.second - span.first); + pos = span.second; + } + + std::string tail = line.substr(pos); + transform(tail); + result += tail; + + line = result; +} + +// ----------------------------------------------------------------------------- + +} // namespace maddy diff --git a/include/maddy/emphasizedparser.h b/include/maddy/emphasizedparser.h index 377e674..170f8ca 100644 --- a/include/maddy/emphasizedparser.h +++ b/include/maddy/emphasizedparser.h @@ -9,6 +9,7 @@ #include #include +#include "maddy/codespanutils.h" #include "maddy/lineparser.h" // ----------------------------------------------------------------------------- @@ -43,12 +44,12 @@ class EmphasizedParser : public LineParser // The leading and trailing `(_*)` groups absorb any leftover underscores // from an unbalanced run (e.g. `__foo_` or `_foo____`), re-emitted // outside the tag instead of into its content. - static std::regex re( - R"((?!.*`.*|.*.*)\b(_*)_(?![\s_])(?!.*`.*|.*<\/code>.*)(.*?[^\s])_(_*)\b(?!.*`.*|.*<\/code>.*))" - ); + static std::regex re(R"(\b(_*)_(?![\s_])(.*?[^\s])_(_*)\b)"); static std::string replacement = "$1$2$3"; - line = std::regex_replace(line, re, replacement); + ApplyOutsideProtectedSpans(line, [](std::string& segment) { + segment = std::regex_replace(segment, re, replacement); + }); } }; // class EmphasizedParser diff --git a/include/maddy/italicparser.h b/include/maddy/italicparser.h index 3f7075d..6a7f8ea 100644 --- a/include/maddy/italicparser.h +++ b/include/maddy/italicparser.h @@ -9,6 +9,7 @@ #include #include +#include "maddy/codespanutils.h" #include "maddy/lineparser.h" // ----------------------------------------------------------------------------- @@ -38,11 +39,12 @@ class ItalicParser : public LineParser */ void Parse(std::string& line) override { - static std::regex re( - R"((?!.*`.*|.*.*)\*(?!.*`.*|.*<\/code>.*)([^\*]*)\*(?!.*`.*|.*<\/code>.*))" - ); + static std::regex re(R"(\*([^\*]*)\*)"); static std::string replacement = "$1"; - line = std::regex_replace(line, re, replacement); + + ApplyOutsideProtectedSpans(line, [](std::string& segment) { + segment = std::regex_replace(segment, re, replacement); + }); } }; // class ItalicParser diff --git a/include/maddy/strikethroughparser.h b/include/maddy/strikethroughparser.h index effa5d8..1f24186 100644 --- a/include/maddy/strikethroughparser.h +++ b/include/maddy/strikethroughparser.h @@ -9,6 +9,7 @@ #include #include +#include "maddy/codespanutils.h" #include "maddy/lineparser.h" // ----------------------------------------------------------------------------- @@ -38,12 +39,12 @@ class StrikeThroughParser : public LineParser */ void Parse(std::string& line) override { - static std::regex re( - R"((?!.*`.*|.*.*)\~\~(?!.*`.*|.*<\/code>.*)([^\~]*)\~\~(?!.*`.*|.*<\/code>.*))" - ); + static std::regex re(R"(\~\~([^\~]*)\~\~)"); static std::string replacement = "$1"; - line = std::regex_replace(line, re, replacement); + ApplyOutsideProtectedSpans(line, [](std::string& segment) { + segment = std::regex_replace(segment, re, replacement); + }); } }; // class StrikeThroughParser diff --git a/include/maddy/strongparser.h b/include/maddy/strongparser.h index b6c0de7..6ad82b2 100644 --- a/include/maddy/strongparser.h +++ b/include/maddy/strongparser.h @@ -9,6 +9,7 @@ #include #include +#include "maddy/codespanutils.h" #include "maddy/lineparser.h" // ----------------------------------------------------------------------------- @@ -43,18 +44,17 @@ class StrongParser : public LineParser // `*` is not a word character, so `\b` next to it does not mean // "edge of a delimiter run" the way it does for `_`; the asterisk // variant is left without a word-boundary anchor. - static std::regex reAsterisk{ - R"((?!.*`.*|.*.*)\*\*(?![\s])(?!.*`.*|.*<\/code>.*)(.*?[^\s])\*\*(?!.*`.*|.*<\/code>.*))" - }; + static std::regex reAsterisk{R"(\*\*(?![\s])(.*?[^\s])\*\*)"}; // The leading and trailing `(_*)` groups absorb any leftover underscores // from an unbalanced run on either side (e.g. `___text__` or // `__text_______`), re-emitted outside the tag by the caller // instead of being swallowed into its content. - static std::regex reUnderscore{ - R"((?!.*`.*|.*.*)\b(_*)__(?![\s_])(?!.*`.*|.*<\/code>.*)(.*?[^\s])__(_*)\b(?!.*`.*|.*<\/code>.*))" - }; - line = std::regex_replace(line, reAsterisk, "$1"); - line = std::regex_replace(line, reUnderscore, "$1$2$3"); + static std::regex reUnderscore{R"(\b(_*)__(?![\s_])(.*?[^\s])__(_*)\b)"}; + + ApplyOutsideProtectedSpans(line, [](std::string& segment) { + segment = std::regex_replace(segment, reAsterisk, "$1"); + segment = std::regex_replace(segment, reUnderscore, "$1$2$3"); + }); } }; // class StrongParser diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index ce5d729..01b0843 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -27,7 +27,7 @@ FetchContent_MakeAvailable(googletest) # ------------------------------------------------------------------------------ -file(GLOB_RECURSE MADDY_TESTS_FILES ${CMAKE_CURRENT_SOURCE_DIR}/maddy/*.cpp) +file(GLOB_RECURSE MADDY_TESTS_FILES ${CMAKE_CURRENT_SOURCE_DIR}/maddy/*.cpp ${CMAKE_CURRENT_SOURCE_DIR}/maddy/*.h ${CMAKE_CURRENT_SOURCE_DIR}/../include/*.h) # ------------------------------------------------------------------------------ diff --git a/tests/maddy/test_maddy_emphasizedparser.cpp b/tests/maddy/test_maddy_emphasizedparser.cpp index 9e1cd52..e4dce08 100644 --- a/tests/maddy/test_maddy_emphasizedparser.cpp +++ b/tests/maddy/test_maddy_emphasizedparser.cpp @@ -177,3 +177,38 @@ TEST(MADDY_EMPHASIZEDPARSER, ItParsesOutsideTickBlocks) ASSERT_EQ(expected, text); } + +// The following cases are adapted from the CommonMark spec +// (https://spec.commonmark.org/), which defines how a code span's +// backtick delimiters are matched and how it interacts with surrounding +// markup. + +TEST(MADDY_EMPHASIZEDPARSER, ItMatchesBacktickRunsByEqualLength) +{ + // CommonMark spec example 349: "`foo``bar``" -> "`foobar". + // The lone opening backtick has no closing run of length 1 (the next + // runs are length 2), so it is ordinary text; the two length-2 runs + // pair up into the code span. Emphasized text on either side of this + // is still parsed normally. + std::string text = "_pre_ `foo``bar`` _post_"; + std::string expected = "pre `foo``bar`` post"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_EMPHASIZEDPARSER, ItLetsALongerBacktickFenceProtectAnInnerBacktick) +{ + // CommonMark spec example 329: "`` foo ` bar ``" -> "foo ` bar". + // A double-backtick fence spans across a single backtick in its + // content; text after the fence is still parsed normally. + std::string text = "`` foo ` bar `` and _emph_"; + std::string expected = "`` foo ` bar `` and emph"; + auto emphasizedParser = std::make_shared(); + + emphasizedParser->Parse(text); + + ASSERT_EQ(expected, text); +} diff --git a/tests/maddy/test_maddy_italicparser.cpp b/tests/maddy/test_maddy_italicparser.cpp index 03c5aac..785757b 100644 --- a/tests/maddy/test_maddy_italicparser.cpp +++ b/tests/maddy/test_maddy_italicparser.cpp @@ -20,3 +20,38 @@ TEST(MADDY_ITALICPARSER, ItReplacesMarkdownWithItalicHTML) ASSERT_EQ(text, expected); } + +// The following cases are adapted from the CommonMark spec +// (https://spec.commonmark.org/), which defines how a code span's +// backtick delimiters are matched and how it interacts with surrounding +// markup. + +TEST(MADDY_ITALICPARSER, ItMatchesBacktickRunsByEqualLength) +{ + // CommonMark spec example 349: "`foo``bar``" -> "`foobar". + // The lone opening backtick has no closing run of length 1 (the next + // runs are length 2), so it is ordinary text; the two length-2 runs + // pair up into the code span. Italic text on either side of this is + // still parsed normally. + std::string text = "*pre* `foo``bar`` *post*"; + std::string expected = "pre `foo``bar`` post"; + auto italicParser = std::make_shared(); + + italicParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_ITALICPARSER, ItLetsALongerBacktickFenceProtectAnInnerBacktick) +{ + // CommonMark spec example 329: "`` foo ` bar ``" -> "foo ` bar". + // A double-backtick fence spans across a single backtick in its + // content; text after the fence is still parsed normally. + std::string text = "`` foo ` bar `` and *italic*"; + std::string expected = "`` foo ` bar `` and italic"; + auto italicParser = std::make_shared(); + + italicParser->Parse(text); + + ASSERT_EQ(expected, text); +} diff --git a/tests/maddy/test_maddy_strikethroughparser.cpp b/tests/maddy/test_maddy_strikethroughparser.cpp index a31108c..cabb935 100644 --- a/tests/maddy/test_maddy_strikethroughparser.cpp +++ b/tests/maddy/test_maddy_strikethroughparser.cpp @@ -33,3 +33,38 @@ TEST(MADDY_STRIKETHROUGHPARSER, ItDoesNotParseInsideInlineCode) ASSERT_EQ(expected, text); } + +// The following cases are adapted from the CommonMark spec +// (https://spec.commonmark.org/), which defines how a code span's +// backtick delimiters are matched and how it interacts with surrounding +// markup. + +TEST(MADDY_STRIKETHROUGHPARSER, ItMatchesBacktickRunsByEqualLength) +{ + // CommonMark spec example 349: "`foo``bar``" -> "`foobar". + // The lone opening backtick has no closing run of length 1 (the next + // runs are length 2), so it is ordinary text; the two length-2 runs + // pair up into the code span. Struck-through text on either side of + // this is still parsed normally. + std::string text = "~~pre~~ `foo``bar`` ~~post~~"; + std::string expected = "pre `foo``bar`` post"; + auto strikeThroughParser = std::make_shared(); + + strikeThroughParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_STRIKETHROUGHPARSER, ItLetsALongerBacktickFenceProtectAnInnerBacktick) +{ + // CommonMark spec example 329: "`` foo ` bar ``" -> "foo ` bar". + // A double-backtick fence spans across a single backtick in its + // content; text after the fence is still parsed normally. + std::string text = "`` foo ` bar `` and ~~struck~~"; + std::string expected = "`` foo ` bar `` and struck"; + auto strikeThroughParser = std::make_shared(); + + strikeThroughParser->Parse(text); + + ASSERT_EQ(expected, text); +} diff --git a/tests/maddy/test_maddy_strongparser.cpp b/tests/maddy/test_maddy_strongparser.cpp index 1211b0e..9a17880 100644 --- a/tests/maddy/test_maddy_strongparser.cpp +++ b/tests/maddy/test_maddy_strongparser.cpp @@ -68,8 +68,10 @@ TEST(MADDY_STRONGPARSER, ItDoesNotParseInsideInlineCode) std::vector tests{ { + // Per CommonMark, a code span protects only its own extent: text + // before it (here, "**bla**") is still eligible for parsing. "some text **bla** `/**text**/` testing `**it**` out", - "some text **bla** `/**text**/` testing `**it**` out", + "some text bla `/**text**/` testing `**it**` out", }, {"some text _bla_ text testing __it__ out", "some text _bla_ text testing it out"}, @@ -84,6 +86,41 @@ TEST(MADDY_STRONGPARSER, ItDoesNotParseInsideInlineCode) } } +// The following cases are adapted from the CommonMark spec +// (https://spec.commonmark.org/), which defines how a code span's +// backtick delimiters are matched and how it interacts with surrounding +// markup. + +TEST(MADDY_STRONGPARSER, ItMatchesBacktickRunsByEqualLength) +{ + // CommonMark spec example 349: "`foo``bar``" -> "`foobar". + // The lone opening backtick has no closing run of length 1 (the next + // runs are length 2), so it is ordinary text; the two length-2 runs + // pair up into the code span. Bold text on either side of this is + // still parsed normally. + std::string text = "**pre** `foo``bar`` **post**"; + std::string expected = "pre `foo``bar`` post"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + +TEST(MADDY_STRONGPARSER, ItLetsALongerBacktickFenceProtectAnInnerBacktick) +{ + // CommonMark spec example 329: "`` foo ` bar ``" -> "foo ` bar". + // A double-backtick fence spans across a single backtick in its + // content; text after the fence is still parsed normally. + std::string text = "`` foo ` bar `` and **bold**"; + std::string expected = "`` foo ` bar `` and bold"; + auto strongParser = std::make_shared(); + + strongParser->Parse(text); + + ASSERT_EQ(expected, text); +} + TEST(MADDY_STRONGPARSER, ItReplacesUnderscoresAtStringEdges) { std::string text = "__some text__"; From 38d3f54b6540fd2b57f8bbd26f36be518a73dbc8 Mon Sep 17 00:00:00 2001 From: Lucian Smith Date: Fri, 18 Sep 2026 08:32:03 -0700 Subject: [PATCH 13/13] Clang formatting fixes. --- include/maddy/codespanutils.h | 18 +++++++++++++----- include/maddy/emphasizedparser.h | 8 +++++--- include/maddy/italicparser.h | 8 +++++--- include/maddy/strikethroughparser.h | 8 +++++--- include/maddy/strongparser.h | 14 ++++++++++---- tests/maddy/test_maddy_strikethroughparser.cpp | 4 +++- tests/maddy/test_maddy_strongparser.cpp | 3 ++- 7 files changed, 43 insertions(+), 20 deletions(-) diff --git a/include/maddy/codespanutils.h b/include/maddy/codespanutils.h index 4e33a04..a4fdd65 100644 --- a/include/maddy/codespanutils.h +++ b/include/maddy/codespanutils.h @@ -59,17 +59,26 @@ inline std::vector> FindProtectedSpans( if (line[i] == '`') { std::size_t runStart = i; - while (i < line.size() && line[i] == '`') { ++i; } + while (i < line.size() && line[i] == '`') + { + ++i; + } std::size_t runLength = i - runStart; std::size_t searchPos = i; while (searchPos < line.size()) { std::size_t closeStart = line.find('`', searchPos); - if (closeStart == std::string::npos) { break; } + if (closeStart == std::string::npos) + { + break; + } std::size_t closeEnd = closeStart; - while (closeEnd < line.size() && line[closeEnd] == '`') { ++closeEnd; } + while (closeEnd < line.size() && line[closeEnd] == '`') + { + ++closeEnd; + } if (closeEnd - closeStart == runLength) { @@ -103,8 +112,7 @@ inline std::vector> FindProtectedSpans( * @return {void} */ inline void ApplyOutsideProtectedSpans( - std::string& line, - const std::function& transform + std::string& line, const std::function& transform ) { std::string result; diff --git a/include/maddy/emphasizedparser.h b/include/maddy/emphasizedparser.h index 170f8ca..4743624 100644 --- a/include/maddy/emphasizedparser.h +++ b/include/maddy/emphasizedparser.h @@ -47,9 +47,11 @@ class EmphasizedParser : public LineParser static std::regex re(R"(\b(_*)_(?![\s_])(.*?[^\s])_(_*)\b)"); static std::string replacement = "$1$2$3"; - ApplyOutsideProtectedSpans(line, [](std::string& segment) { - segment = std::regex_replace(segment, re, replacement); - }); + ApplyOutsideProtectedSpans( + line, + [](std::string& segment) + { segment = std::regex_replace(segment, re, replacement); } + ); } }; // class EmphasizedParser diff --git a/include/maddy/italicparser.h b/include/maddy/italicparser.h index 6a7f8ea..1b21e03 100644 --- a/include/maddy/italicparser.h +++ b/include/maddy/italicparser.h @@ -42,9 +42,11 @@ class ItalicParser : public LineParser static std::regex re(R"(\*([^\*]*)\*)"); static std::string replacement = "$1"; - ApplyOutsideProtectedSpans(line, [](std::string& segment) { - segment = std::regex_replace(segment, re, replacement); - }); + ApplyOutsideProtectedSpans( + line, + [](std::string& segment) + { segment = std::regex_replace(segment, re, replacement); } + ); } }; // class ItalicParser diff --git a/include/maddy/strikethroughparser.h b/include/maddy/strikethroughparser.h index 1f24186..df18a9c 100644 --- a/include/maddy/strikethroughparser.h +++ b/include/maddy/strikethroughparser.h @@ -42,9 +42,11 @@ class StrikeThroughParser : public LineParser static std::regex re(R"(\~\~([^\~]*)\~\~)"); static std::string replacement = "$1"; - ApplyOutsideProtectedSpans(line, [](std::string& segment) { - segment = std::regex_replace(segment, re, replacement); - }); + ApplyOutsideProtectedSpans( + line, + [](std::string& segment) + { segment = std::regex_replace(segment, re, replacement); } + ); } }; // class StrikeThroughParser diff --git a/include/maddy/strongparser.h b/include/maddy/strongparser.h index 6ad82b2..5e39223 100644 --- a/include/maddy/strongparser.h +++ b/include/maddy/strongparser.h @@ -51,10 +51,16 @@ class StrongParser : public LineParser // instead of being swallowed into its content. static std::regex reUnderscore{R"(\b(_*)__(?![\s_])(.*?[^\s])__(_*)\b)"}; - ApplyOutsideProtectedSpans(line, [](std::string& segment) { - segment = std::regex_replace(segment, reAsterisk, "$1"); - segment = std::regex_replace(segment, reUnderscore, "$1$2$3"); - }); + ApplyOutsideProtectedSpans( + line, + [](std::string& segment) + { + segment = + std::regex_replace(segment, reAsterisk, "$1"); + segment = + std::regex_replace(segment, reUnderscore, "$1$2$3"); + } + ); } }; // class StrongParser diff --git a/tests/maddy/test_maddy_strikethroughparser.cpp b/tests/maddy/test_maddy_strikethroughparser.cpp index cabb935..57c7736 100644 --- a/tests/maddy/test_maddy_strikethroughparser.cpp +++ b/tests/maddy/test_maddy_strikethroughparser.cpp @@ -55,7 +55,9 @@ TEST(MADDY_STRIKETHROUGHPARSER, ItMatchesBacktickRunsByEqualLength) ASSERT_EQ(expected, text); } -TEST(MADDY_STRIKETHROUGHPARSER, ItLetsALongerBacktickFenceProtectAnInnerBacktick) +TEST( + MADDY_STRIKETHROUGHPARSER, ItLetsALongerBacktickFenceProtectAnInnerBacktick +) { // CommonMark spec example 329: "`` foo ` bar ``" -> "foo ` bar". // A double-backtick fence spans across a single backtick in its diff --git a/tests/maddy/test_maddy_strongparser.cpp b/tests/maddy/test_maddy_strongparser.cpp index 9a17880..48675f8 100644 --- a/tests/maddy/test_maddy_strongparser.cpp +++ b/tests/maddy/test_maddy_strongparser.cpp @@ -99,7 +99,8 @@ TEST(MADDY_STRONGPARSER, ItMatchesBacktickRunsByEqualLength) // pair up into the code span. Bold text on either side of this is // still parsed normally. std::string text = "**pre** `foo``bar`` **post**"; - std::string expected = "pre `foo``bar`` post"; + std::string expected = + "pre `foo``bar`` post"; auto strongParser = std::make_shared(); strongParser->Parse(text);