Files
Elternbeirat/Elternbeirat.Web/Shared/Markdown.cs
T

245 lines
11 KiB
C#

using System.Text;
using System.Text.RegularExpressions;
using Markdig;
using Markdig.Renderers.Html;
using Markdig.Syntax;
using Markdig.Syntax.Inlines;
namespace Elternbeirat.Web.Shared;
/// <summary>
/// Renders the editors' Markdown (the <c>body</c>, <c>intro</c> and <c>answer</c>
/// fields from PocketBase) to the HTML the pages show.
/// </summary>
/// <remarks>
/// The pipeline is CommonMark plus exactly two extensions, each there for a reason
/// an editor can see:
/// <list type="bullet">
/// <item>
/// <description>
/// <b>Custom containers</b> (<c>::: name</c> … <c>:::</c>) turn a block
/// into <c>&lt;div class="name"&gt;</c>. That is how editors use the
/// design blocks (<c>kennzahlen</c>, <c>aufruf</c>, <c>kacheln</c>,
/// <c>hinweis</c>) without writing HTML. An unknown name just yields a
/// div without styling, so a typo never breaks a page.
/// </description>
/// </item>
/// <item>
/// <description>
/// <b>Pipe tables</b>: plain CommonMark has no tables at all, so without
/// this the table styles in <c>app.css</c> could never apply.
/// </description>
/// </item>
/// </list>
/// <para>
/// Raw HTML in the Markdown is <b>escaped</b>, not passed through
/// (<see cref="MarkdownPipelineBuilder"/>'s <c>DisableHtml</c>). The result is
/// rendered as a <c>MarkupString</c>, so a passed-through <c>&lt;script&gt;</c>
/// would run in the visitor's browser, and the privacy text promises that no
/// program code runs there. Editors style pages with the blocks instead.
/// </para>
/// <para>
/// After parsing, a paragraph (or list item) that consists of nothing but one
/// link gets the class <see cref="LoneLinkClass"/>. Pure CSS cannot tell
/// "a link alone on its line" from "a link inside a sentence" (selectors ignore
/// the text around an element), but the stylesheet needs exactly that to show a
/// lone <c>mailto:</c> link as a button and a lone PDF link as a file card.
/// </para>
/// </remarks>
/// <seealso cref="IconSet"/>
public static partial class Markdown
{
/// <summary>
/// The class marking a paragraph or list item whose only content is one link.
/// <c>app.css</c> keys the mail button and the PDF card off it.
/// </summary>
public const string LoneLinkClass = "lone-link";
private static readonly MarkdownPipeline Pipeline =
new MarkdownPipelineBuilder()
.UseCustomContainers()
.UsePipeTables()
.DisableHtml()
.Build();
/// <summary>
/// Converts a Markdown string to HTML.
/// </summary>
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
/// <returns>
/// The rendered HTML, or an empty string for <see langword="null"/> or empty
/// input, so callers can bind the result directly.
/// </returns>
/// <example>
/// <code>
/// Markdown.ToHtml("::: hinweis\nBitte vormerken.\n:::");
/// // &lt;div class="hinweis"&gt;&lt;p&gt;Bitte vormerken.&lt;/p&gt;&lt;/div&gt;
/// </code>
/// </example>
public static string ToHtml(string? markdown) =>
string.IsNullOrEmpty(markdown) ? "" : Render(Parse(markdown));
/// <summary>
/// Extracts the first sentence of a Markdown text as plain text, e.g. as the
/// short description on a home page tile.
/// </summary>
/// <param name="markdown">The editor's Markdown; may be <see langword="null"/>.</param>
/// <returns>
/// The first sentence of the first top-level paragraph, with all Markdown
/// removed, whitespace collapsed and a trailing colon dropped, or an empty
/// string when there is no such paragraph (empty body, or only headings, lists
/// and blocks).
/// </returns>
/// <remarks>
/// Only top-level paragraphs count: a heading repeats the title, and the text in
/// a list or a <c>:::</c> block is rarely a sentence that describes the page.
/// <para>
/// A sentence ends at <c>.</c>, <c>!</c> or <c>?</c> followed by a space or
/// the end. A period after a single letter or a number does not count, so
/// <c>z. B.</c> and <c>13. November</c> do not cut the sentence short. A
/// longer abbreviation such as <c>bzw.</c> still does; that costs the rest
/// of a teaser, never the page.
/// </para>
/// <para>
/// A paragraph that leads into a list often ends in <c>:</c> without a
/// period; as a teaser the colon would point at a list that is not there.
/// The sentence is returned whole: a long one is cut by the tile's CSS,
/// which knows the space it has, not by a character count.
/// </para>
/// </remarks>
/// <example>
/// <code>
/// Markdown.FirstSentence("# Vorstand\n\nWir sind **sieben** Eltern. Mehr unten.");
/// // "Wir sind sieben Eltern."
/// </code>
/// </example>
public static string FirstSentence(string? markdown) =>
string.IsNullOrWhiteSpace(markdown)
? ""
: Parse(markdown).OfType<ParagraphBlock>().FirstOrDefault()?.Inline is { } inline
? UpToSentenceEnd(PlainText(inline)).TrimEnd(':', ' ')
: "";
/// <summary>
/// Parses Markdown with the site's pipeline, for callers that take the document
/// apart before rendering it (see <see cref="Render"/>).
/// </summary>
/// <param name="markdown">The editor's Markdown.</param>
/// <returns>The parsed document.</returns>
internal static MarkdownDocument Parse(string markdown) =>
Markdig.Markdown.Parse(markdown, Pipeline);
/// <summary>
/// Renders a document from <see cref="Parse"/> to HTML, marking lone links on
/// the way, exactly as <see cref="ToHtml"/> does.
/// </summary>
/// <param name="document">The parsed document; changed in place.</param>
/// <returns>The rendered HTML.</returns>
internal static string Render(MarkdownDocument document) =>
Markdig.Markdown.ToHtml(MarkLoneLinks(document), Pipeline);
/// <summary>
/// Flattens inline Markdown to the text a reader sees: emphasis and link
/// markup dropped, link text kept, images left out, whitespace collapsed.
/// </summary>
/// <param name="inline">The inline content, e.g. of a paragraph or heading.</param>
/// <returns>The visible text, trimmed.</returns>
internal static string PlainText(ContainerInline inline) =>
Whitespace().Replace(AppendText(new StringBuilder(), inline).ToString(), " ").Trim();
/// <summary>
/// Whether an inline carries no visible content (whitespace or a line break).
/// </summary>
/// <param name="inline">The inline to inspect.</param>
/// <returns>
/// <see langword="true"/> if it can be ignored when looking for links that stand
/// alone in a paragraph.
/// </returns>
internal static bool IsBlank(Inline inline) =>
inline switch
{
LineBreakInline => true,
LiteralInline literal => literal.Content.IsEmptyOrWhitespace(),
_ => false,
};
/// <summary>
/// Appends the visible text of an inline and its children.
/// </summary>
/// <param name="text">The builder to append to.</param>
/// <param name="inline">The inline to flatten.</param>
/// <returns>The same <paramref name="text"/>, for chaining.</returns>
private static StringBuilder AppendText(StringBuilder text, Inline inline) =>
inline switch
{
LiteralInline literal => text.Append(literal.Content.ToString()),
CodeInline code => text.Append(code.Content),
HtmlEntityInline entity => text.Append(entity.Transcoded.ToString()),
AutolinkInline autolink => text.Append(autolink.Url),
LineBreakInline => text.Append(' '),
LinkInline { IsImage: true } => text,
ContainerInline container => container.Aggregate(text, AppendText),
_ => text,
};
/// <summary>
/// Cuts a text after its first sentence.
/// </summary>
/// <param name="text">Plain text.</param>
/// <returns>
/// The first sentence including its end mark, or the whole text if it has none.
/// </returns>
private static string UpToSentenceEnd(string text) =>
SentenceEnd().Match(text) is { Success: true } end ? text[..(end.Index + 1)] : text;
/// <summary>
/// Matches the end mark of a sentence: <c>.</c>, <c>!</c> or <c>?</c> before a
/// space or the end, unless it follows a lone letter (<c>z.</c>) or a number
/// (<c>13.</c>).
/// </summary>
[GeneratedRegex(@"(?<!\b\p{L}|\b\d+)[.!?](?=\s|$)")]
private static partial Regex SentenceEnd();
/// <summary>
/// Matches a run of whitespace, collapsed to one space in plain text.
/// </summary>
[GeneratedRegex(@"\s+")]
private static partial Regex Whitespace();
/// <summary>
/// Adds <see cref="LoneLinkClass"/> to every paragraph that holds nothing but
/// one link.
/// </summary>
/// <param name="document">The parsed document; changed in place.</param>
/// <returns>The same <paramref name="document"/>, for chaining.</returns>
/// <remarks>
/// Inside a list item the class goes on the item, not the paragraph: in a tight
/// list Markdig writes no <c>&lt;p&gt;</c> at all, so a class on the paragraph
/// would be silently dropped.
/// </remarks>
private static MarkdownDocument MarkLoneLinks(MarkdownDocument document)
{
foreach (var paragraph in document.Descendants<ParagraphBlock>().Where(IsLoneLink))
{
MarkdownObject target = paragraph.Parent is ListItemBlock { Count: 1 } item ? item : paragraph;
target.GetAttributes().AddClass(LoneLinkClass);
}
return document;
}
/// <summary>
/// Whether a paragraph's content is exactly one link, ignoring surrounding
/// whitespace and line breaks.
/// </summary>
/// <param name="paragraph">The paragraph to inspect.</param>
/// <returns>
/// <see langword="true"/> for one link (inline <c>[text](url)</c> or an
/// autolink <c>&lt;url&gt;</c>, but not an image) and nothing else.
/// </returns>
private static bool IsLoneLink(ParagraphBlock paragraph) =>
paragraph.Inline?
.Where(inline => !IsBlank(inline))
.ToList() is [LinkInline { IsImage: false } or AutolinkInline];
}