using System.Text; using System.Text.RegularExpressions; using Markdig; using Markdig.Renderers.Html; using Markdig.Syntax; using Markdig.Syntax.Inlines; namespace Elternbeirat.Web.Shared; /// /// Renders the editors' Markdown (the body, intro and answer /// fields from PocketBase) to the HTML the pages show. /// /// /// The pipeline is CommonMark plus exactly two extensions, each there for a reason /// an editor can see: /// /// /// /// Custom containers (::: name … :::) turn a block /// into <div class="name">. That is how editors use the /// design blocks (kennzahlen, aufruf, kacheln, /// hinweis) without writing HTML. An unknown name just yields a /// div without styling, so a typo never breaks a page. /// /// /// /// /// Pipe tables: plain CommonMark has no tables at all, so without /// this the table styles in app.css could never apply. /// /// /// /// /// Raw HTML in the Markdown is escaped, not passed through /// ('s DisableHtml). The result is /// rendered as a MarkupString, so a passed-through <script> /// would run in the visitor's browser, and the privacy text promises that no /// program code runs there. Editors style pages with the blocks instead. /// /// /// After parsing, a paragraph (or list item) that consists of nothing but one /// link gets the class . Pure CSS cannot tell /// "a link alone on its line" from "a link inside a sentence" (selectors ignore /// the text around an element), but the stylesheet needs exactly that to show a /// lone mailto: link as a button and a lone PDF link as a file card. /// /// /// public static partial class Markdown { /// /// The class marking a paragraph or list item whose only content is one link. /// app.css keys the mail button and the PDF card off it. /// public const string LoneLinkClass = "lone-link"; /// /// The default teaser length: about two lines on a card, enough to say what a /// post is about without turning the list into a wall of text. /// public const int TeaserLength = 160; private static readonly MarkdownPipeline Pipeline = new MarkdownPipelineBuilder() .UseCustomContainers() .UsePipeTables() .DisableHtml() .Build(); /// /// Converts a Markdown string to HTML. /// /// The editor's Markdown; may be . /// /// The rendered HTML, or an empty string for or empty /// input, so callers can bind the result directly. /// /// /// /// Markdown.ToHtml("::: hinweis\nBitte vormerken.\n:::"); /// // <div class="hinweis"><p>Bitte vormerken.</p></div> /// /// public static string ToHtml(string? markdown) => string.IsNullOrEmpty(markdown) ? "" : Render(Parse(markdown)); /// /// Extracts the first sentence of a Markdown text as plain text, e.g. as the /// short description on a home page tile. /// /// The editor's Markdown; may be . /// /// The first sentence of the first top-level paragraph, with all Markdown /// removed, whitespace collapsed and a trailing colon dropped, or an empty /// string when there is no such paragraph (empty body, or only headings, lists /// and blocks). /// /// /// Only top-level paragraphs count: a heading repeats the title, and the text in /// a list or a ::: block is rarely a sentence that describes the page. /// /// A sentence ends at ., ! or ? followed by a space or /// the end. A period after a single letter or a number does not count, so /// z. B. and 13. November do not cut the sentence short. A /// longer abbreviation such as bzw. still does; that costs the rest /// of a teaser, never the page. /// /// /// A paragraph that leads into a list often ends in : without a /// period; as a teaser the colon would point at a list that is not there. /// The sentence is returned whole: a long one is cut by the tile's CSS, /// which knows the space it has, not by a character count. /// /// /// /// /// Markdown.FirstSentence("# Vorstand\n\nWir sind **sieben** Eltern. Mehr unten."); /// // "Wir sind sieben Eltern." /// /// public static string FirstSentence(string? markdown) => string.IsNullOrWhiteSpace(markdown) ? "" : Parse(markdown).OfType().FirstOrDefault()?.Inline is { } inline ? UpToSentenceEnd(PlainText(inline)).TrimEnd(':', ' ') : ""; /// /// Builds a short plain-text teaser from a Markdown text, e.g. for a post card. /// /// The editor's Markdown; may be . /// /// The longest teaser, in characters, before the ellipsis. Defaults to /// . /// /// /// The text of all paragraphs (including those in lists, quotes and ::: /// blocks), with all Markdown removed and whitespace collapsed. A longer text is /// cut at the last word boundary within and ends /// in …. An empty string when there is no paragraph text at all. /// /// /// is zero or negative. /// /// /// Headings are skipped: on a card the title already heads the teaser. Unlike /// the teaser does not stop at the first sentence, /// because a post often opens with a short line ("Liebe Eltern,") that says /// nothing on its own. /// /// Punctuation left dangling at the cut (,, ., : …) is /// dropped, so the teaser never ends in ,…. /// /// /// /// /// Markdown.Teaser("## Rückblick\n\nDer **Basar** war ein voller Erfolg.", 20); /// // "Der Basar war ein…" /// /// public static string Teaser(string? markdown, int maxLength = TeaserLength) => maxLength <= 0 ? throw new ArgumentOutOfRangeException(nameof(maxLength), maxLength, "The teaser needs room for at least one character.") : string.IsNullOrWhiteSpace(markdown) ? "" : Shorten( string.Join( ' ', Parse(markdown) .Descendants() .Select(paragraph => paragraph.Inline is { } inline ? PlainText(inline) : "") .Where(text => text.Length > 0)), maxLength); /// /// Parses Markdown with the site's pipeline, for callers that take the document /// apart before rendering it (see ). /// /// The editor's Markdown. /// The parsed document. internal static MarkdownDocument Parse(string markdown) => Markdig.Markdown.Parse(markdown, Pipeline); /// /// Renders a document from to HTML, marking lone links on /// the way, exactly as does. /// /// The parsed document; changed in place. /// The rendered HTML. internal static string Render(MarkdownDocument document) => Markdig.Markdown.ToHtml(MarkLoneLinks(document), Pipeline); /// /// Flattens inline Markdown to the text a reader sees: emphasis and link /// markup dropped, link text kept, images left out, whitespace collapsed. /// /// The inline content, e.g. of a paragraph or heading. /// The visible text, trimmed. internal static string PlainText(ContainerInline inline) => Whitespace().Replace(AppendText(new StringBuilder(), inline).ToString(), " ").Trim(); /// /// Whether an inline carries no visible content (whitespace or a line break). /// /// The inline to inspect. /// /// if it can be ignored when looking for links that stand /// alone in a paragraph. /// internal static bool IsBlank(Inline inline) => inline switch { LineBreakInline => true, LiteralInline literal => literal.Content.IsEmptyOrWhitespace(), _ => false, }; /// /// Appends the visible text of an inline and its children. /// /// The builder to append to. /// The inline to flatten. /// The same , for chaining. private static StringBuilder AppendText(StringBuilder text, Inline inline) => inline switch { LiteralInline literal => text.Append(literal.Content.ToString()), CodeInline code => text.Append(code.Content), HtmlEntityInline entity => text.Append(entity.Transcoded.ToString()), AutolinkInline autolink => text.Append(autolink.Url), LineBreakInline => text.Append(' '), LinkInline { IsImage: true } => text, ContainerInline container => container.Aggregate(text, AppendText), _ => text, }; /// /// Cuts a text after its first sentence. /// /// Plain text. /// /// The first sentence including its end mark, or the whole text if it has none. /// private static string UpToSentenceEnd(string text) => SentenceEnd().Match(text) is { Success: true } end ? text[..(end.Index + 1)] : text; /// /// Cuts a text to a word boundary and marks the cut. /// /// Plain text. /// The longest result before the ellipsis. /// /// The text unchanged if it fits; otherwise everything up to the last space /// within , trailing punctuation removed, plus /// …. A single word longer than is cut hard. /// /// /// The space is searched one character past the limit: if the text breaks /// exactly at , the last whole word still fits. /// private static string Shorten(string text, int maxLength) => text.Length <= maxLength ? text : (text[..(maxLength + 1)].LastIndexOf(' ') is var space and > 0 ? text[..space] : text[..maxLength]) .TrimEnd(' ', ',', ';', ':', '.', '-', '–') + "…"; /// /// Matches the end mark of a sentence: ., ! or ? before a /// space or the end, unless it follows a lone letter (z.) or a number /// (13.). /// [GeneratedRegex(@"(? /// Matches a run of whitespace, collapsed to one space in plain text. /// [GeneratedRegex(@"\s+")] private static partial Regex Whitespace(); /// /// Adds to every paragraph that holds nothing but /// one link. /// /// The parsed document; changed in place. /// The same , for chaining. /// /// Inside a list item the class goes on the item, not the paragraph: in a tight /// list Markdig writes no <p> at all, so a class on the paragraph /// would be silently dropped. /// private static MarkdownDocument MarkLoneLinks(MarkdownDocument document) { foreach (var paragraph in document.Descendants().Where(IsLoneLink)) { MarkdownObject target = paragraph.Parent is ListItemBlock { Count: 1 } item ? item : paragraph; target.GetAttributes().AddClass(LoneLinkClass); } return document; } /// /// Whether a paragraph's content is exactly one link, ignoring surrounding /// whitespace and line breaks. /// /// The paragraph to inspect. /// /// for one link (inline [text](url) or an /// autolink <url>, but not an image) and nothing else. /// private static bool IsLoneLink(ParagraphBlock paragraph) => paragraph.Inline? .Where(inline => !IsBlank(inline)) .ToList() is [LinkInline { IsImage: false } or AutolinkInline]; }